@yeaft/webchat-agent 1.0.446 → 1.0.448
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/local-runtime/server/context.js +3 -22
- package/local-runtime/server/database.js +1 -0
- package/local-runtime/server/db/agent-inventory-db.js +122 -0
- package/local-runtime/server/db/connection.js +59 -0
- package/local-runtime/server/handlers/agent-sync.js +12 -1
- package/local-runtime/server/handlers/client-conversation.js +1 -0
- package/local-runtime/server/routes/admin-routes.js +78 -15
- package/local-runtime/server/ws-agent.js +41 -3
- package/local-runtime/server/ws-client.js +1 -7
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +208 -233
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/archive/turn-archive.js +1 -1
- package/yeaft/cli.js +15 -17
- package/yeaft/config.js +2 -3
- package/yeaft/conversation/internal-control.js +2 -0
- package/yeaft/conversation/persist.js +22 -216
- package/yeaft/conversation/visible-entry.js +1 -1
- package/yeaft/effort.js +0 -2
- package/yeaft/engine.js +31 -376
- package/yeaft/history-window.js +570 -0
- package/yeaft/llm/adapter.js +1 -1
- package/yeaft/llm/anthropic.js +1 -1
- package/yeaft/llm/models-dev.js +2 -1
- package/yeaft/llm/openai-responses.js +1 -1
- package/yeaft/llm/router.js +2 -3
- package/yeaft/llm/usage-accounting.js +2 -2
- package/yeaft/pair-sanitize.js +3 -3
- package/yeaft/prompts.js +3 -4
- package/yeaft/session.js +0 -51
- package/yeaft/stdio-protocol.js +0 -1
- package/yeaft/stop-hooks.js +5 -6
- package/yeaft/turn-utils.js +5 -5
- package/yeaft/web-bridge.js +57 -135
- package/yeaft/compact/compactor.js +0 -283
- package/yeaft/compact/orchestrator.js +0 -141
- package/yeaft/compact/partition.js +0 -94
- package/yeaft/compact/triggers.js +0 -54
- package/yeaft/compact/turn-group.js +0 -85
- package/yeaft/history-compact.js +0 -764
|
@@ -0,0 +1,570 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* history-window.js — deterministic history shaping for provider requests.
|
|
3
|
+
*
|
|
4
|
+
* This module never calls an LLM, writes a summary, archives transcript rows,
|
|
5
|
+
* or changes the persisted conversation. It builds bounded copies; callers may
|
|
6
|
+
* replace a disposable runtime cache with one of those copies. The persisted
|
|
7
|
+
* message history remains authoritative; Memory/Dream is the long-lived
|
|
8
|
+
* semantic context.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { estimateTokens } from './conversation/persist.js';
|
|
12
|
+
import { pairSanitize } from './pair-sanitize.js';
|
|
13
|
+
import { truncateToolResultIfNeeded } from './tools/registry.js';
|
|
14
|
+
import { countTurns, indexOfNthTurnFromEnd, sliceLastNTurns } from './turn-utils.js';
|
|
15
|
+
|
|
16
|
+
export const DEFAULT_KEEP_TOOL_TURNS = 3;
|
|
17
|
+
export const DEFAULT_RECENT_TURN_CAP = 25;
|
|
18
|
+
export const DEFAULT_MESSAGE_TOKEN_BUDGET = 32768;
|
|
19
|
+
|
|
20
|
+
// Runtime history is a cache, not a second transcript. Keep its hard cap
|
|
21
|
+
// independent from a user-configured provider budget so a large config cannot
|
|
22
|
+
// turn the bridge cache back into an unbounded transcript.
|
|
23
|
+
export const DEFAULT_RUNTIME_CACHE_TURN_CAP = 25;
|
|
24
|
+
export const DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET = 32768;
|
|
25
|
+
export const DEFAULT_RUNTIME_CACHE_MESSAGE_CAP = 256;
|
|
26
|
+
|
|
27
|
+
const IMAGE_PART_TOKEN_COST = 1024;
|
|
28
|
+
const DOCUMENT_PART_TOKEN_COST = 2048;
|
|
29
|
+
const CONTENT_PART_FRAME_TOKENS = 2;
|
|
30
|
+
const TEXT_CHARS_PER_TOKEN = 4;
|
|
31
|
+
const BINARY_CHARS_PER_TOKEN = 16;
|
|
32
|
+
const OVERSIZED_ATTACHMENT_MARKER = '[attachment omitted from provider context budget]';
|
|
33
|
+
|
|
34
|
+
function serializeJsonValue(value) {
|
|
35
|
+
if (typeof value === 'string') return value;
|
|
36
|
+
try {
|
|
37
|
+
const serialized = JSON.stringify(value ?? '');
|
|
38
|
+
return typeof serialized === 'string' ? serialized : String(value ?? '');
|
|
39
|
+
} catch {
|
|
40
|
+
return String(value ?? '');
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function safeJsonTokenEstimate(value) {
|
|
45
|
+
return estimateTokens(serializeJsonValue(value));
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function thinkingBlockWirePart(block) {
|
|
49
|
+
if (!block || typeof block !== 'object') return null;
|
|
50
|
+
if (typeof block.signature !== 'string' || !block.signature) return null;
|
|
51
|
+
if (block.redacted) {
|
|
52
|
+
if (typeof block.data !== 'string') return null;
|
|
53
|
+
return { type: 'redacted_thinking', data: block.data, signature: block.signature };
|
|
54
|
+
}
|
|
55
|
+
if (typeof block.thinking !== 'string') return null;
|
|
56
|
+
return { type: 'thinking', thinking: block.thinking, signature: block.signature };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function validThinkingBlocks(blocks) {
|
|
60
|
+
if (!Array.isArray(blocks)) return [];
|
|
61
|
+
return blocks.filter(block => thinkingBlockWirePart(block));
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function estimateThinkingBlockTokens(block) {
|
|
65
|
+
const part = thinkingBlockWirePart(block);
|
|
66
|
+
return part ? estimateContentPartTokens(part) : safeJsonTokenEstimate(block);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function estimateThinkingBlocksTokens(blocks) {
|
|
70
|
+
if (!Array.isArray(blocks) || blocks.length === 0) return 0;
|
|
71
|
+
return blocks.reduce((total, block) => total + estimateThinkingBlockTokens(block), CONTENT_PART_FRAME_TOKENS);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function binaryPayloadTokenEstimate(value) {
|
|
75
|
+
if (typeof value !== 'string' || value.length === 0) return 0;
|
|
76
|
+
// Base64/image bytes are not text tokens, so do not charge them at the text
|
|
77
|
+
// ratio. Still count a conservative wire/configuration cost; otherwise a
|
|
78
|
+
// huge content part would bypass the request budget entirely.
|
|
79
|
+
return Math.ceil(value.length / BINARY_CHARS_PER_TOKEN);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function partMetadataTokenEstimate(part, fields = []) {
|
|
83
|
+
return fields.reduce((total, field) => {
|
|
84
|
+
const value = part?.[field];
|
|
85
|
+
return total + (typeof value === 'string' ? estimateTokens(value) : 0);
|
|
86
|
+
}, 0);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Estimate one provider content part. This is a guardrail, not a tokenizer;
|
|
91
|
+
* the provider remains authoritative about the actual context limit.
|
|
92
|
+
*
|
|
93
|
+
* @param {unknown} part
|
|
94
|
+
* @returns {number}
|
|
95
|
+
*/
|
|
96
|
+
export function estimateContentPartTokens(part) {
|
|
97
|
+
if (typeof part === 'string') return estimateTokens(part);
|
|
98
|
+
if (!part || typeof part !== 'object') return estimateTokens(String(part ?? ''));
|
|
99
|
+
|
|
100
|
+
const type = String(part.type || '');
|
|
101
|
+
if (type === 'text' || type === 'input_text' || type === 'output_text') {
|
|
102
|
+
return estimateTokens(typeof part.text === 'string' ? part.text : '');
|
|
103
|
+
}
|
|
104
|
+
if (type === 'thinking') {
|
|
105
|
+
return 4 + estimateTokens(part.thinking || '') + estimateTokens(part.signature || '');
|
|
106
|
+
}
|
|
107
|
+
if (type === 'redacted_thinking') {
|
|
108
|
+
return 4 + estimateTokens(part.data || '') + estimateTokens(part.signature || '');
|
|
109
|
+
}
|
|
110
|
+
if (type === 'image' || type === 'input_image') {
|
|
111
|
+
const source = part.source && typeof part.source === 'object' ? part.source : part;
|
|
112
|
+
return IMAGE_PART_TOKEN_COST
|
|
113
|
+
+ partMetadataTokenEstimate(part, ['title', 'alt', 'image_url'])
|
|
114
|
+
+ partMetadataTokenEstimate(source, ['url', 'media_type', 'mediaType'])
|
|
115
|
+
+ binaryPayloadTokenEstimate(source.data);
|
|
116
|
+
}
|
|
117
|
+
if (type === 'document' || type === 'input_file') {
|
|
118
|
+
const source = part.source && typeof part.source === 'object' ? part.source : part;
|
|
119
|
+
return DOCUMENT_PART_TOKEN_COST
|
|
120
|
+
+ partMetadataTokenEstimate(part, ['title', 'filename', 'file_data'])
|
|
121
|
+
+ partMetadataTokenEstimate(source, ['media_type', 'mediaType', 'url'])
|
|
122
|
+
+ binaryPayloadTokenEstimate(source.data);
|
|
123
|
+
}
|
|
124
|
+
if (type === 'tool_result') {
|
|
125
|
+
return 4 + estimateContentTokens(part.content);
|
|
126
|
+
}
|
|
127
|
+
if (type === 'function_call_output') {
|
|
128
|
+
return 4 + estimateTokens(serializeJsonValue(part.output));
|
|
129
|
+
}
|
|
130
|
+
return safeJsonTokenEstimate(part);
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Estimate string or array provider content, including multimodal parts.
|
|
135
|
+
*
|
|
136
|
+
* @param {unknown} content
|
|
137
|
+
* @returns {number}
|
|
138
|
+
*/
|
|
139
|
+
export function estimateContentTokens(content) {
|
|
140
|
+
if (typeof content === 'string') return estimateTokens(content);
|
|
141
|
+
if (Array.isArray(content)) {
|
|
142
|
+
return CONTENT_PART_FRAME_TOKENS
|
|
143
|
+
+ content.reduce((total, part) => total + estimateContentPartTokens(part), 0);
|
|
144
|
+
}
|
|
145
|
+
if (content == null) return 0;
|
|
146
|
+
return safeJsonTokenEstimate(content);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Estimate the provider-token weight of one message.
|
|
151
|
+
*
|
|
152
|
+
* @param {object} message
|
|
153
|
+
* @returns {number}
|
|
154
|
+
*/
|
|
155
|
+
export function estimateMessageTokens(message) {
|
|
156
|
+
if (!message || typeof message !== 'object') return 0;
|
|
157
|
+
let total = 2 + estimateContentTokens(message.content);
|
|
158
|
+
total += estimateThinkingBlocksTokens(message.thinkingBlocks);
|
|
159
|
+
if (Array.isArray(message.toolCalls)) {
|
|
160
|
+
for (const toolCall of message.toolCalls) {
|
|
161
|
+
total += 4;
|
|
162
|
+
try {
|
|
163
|
+
const input = typeof toolCall.input === 'string'
|
|
164
|
+
? toolCall.input
|
|
165
|
+
: JSON.stringify(toolCall.input || {});
|
|
166
|
+
total += estimateTokens(input);
|
|
167
|
+
} catch {
|
|
168
|
+
// Ignore malformed tool input in the approximate guardrail.
|
|
169
|
+
}
|
|
170
|
+
if (toolCall.name) total += estimateTokens(toolCall.name);
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
if (message.toolCallId) total += 2;
|
|
174
|
+
return total;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* @param {Array<object>} messages
|
|
179
|
+
* @returns {number}
|
|
180
|
+
*/
|
|
181
|
+
export function estimateMessagesTokens(messages) {
|
|
182
|
+
if (!Array.isArray(messages)) return 0;
|
|
183
|
+
return messages.reduce((total, message) => total + estimateMessageTokens(message), 0);
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
function hasContentAfterToolStrip(content) {
|
|
187
|
+
if (typeof content === 'string') return content.trim().length > 0;
|
|
188
|
+
if (Array.isArray(content)) {
|
|
189
|
+
return content.some(part => {
|
|
190
|
+
if (typeof part === 'string') return part.trim().length > 0;
|
|
191
|
+
if (!part || typeof part !== 'object') return part != null;
|
|
192
|
+
return typeof part.text === 'string' ? part.text.trim().length > 0 : true;
|
|
193
|
+
});
|
|
194
|
+
}
|
|
195
|
+
return content != null;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function stripToolContentParts(content) {
|
|
199
|
+
if (!Array.isArray(content)) return content;
|
|
200
|
+
return content.filter(part => {
|
|
201
|
+
if (!part || typeof part !== 'object') return true;
|
|
202
|
+
return part.type !== 'tool_use'
|
|
203
|
+
&& part.type !== 'tool_result'
|
|
204
|
+
&& part.type !== 'function_call'
|
|
205
|
+
&& part.type !== 'function_call_output';
|
|
206
|
+
});
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Remove old tool payloads from the provider copy while keeping ordinary user
|
|
211
|
+
* and assistant text. The newest tool turns remain lossless so the active tool
|
|
212
|
+
* protocol stays paired; pairSanitize runs after this transform.
|
|
213
|
+
*
|
|
214
|
+
* @param {Array<object>} messages
|
|
215
|
+
* @param {{ keepToolTurns?: number }} [options]
|
|
216
|
+
* @returns {Array<object>}
|
|
217
|
+
*/
|
|
218
|
+
export function stripToolNoiseFromOlderTurns(messages, options = {}) {
|
|
219
|
+
if (!Array.isArray(messages) || messages.length === 0) return [];
|
|
220
|
+
const keepToolTurns = Number.isFinite(options.keepToolTurns) && options.keepToolTurns >= 0
|
|
221
|
+
? options.keepToolTurns
|
|
222
|
+
: DEFAULT_KEEP_TOOL_TURNS;
|
|
223
|
+
const cutIndex = indexOfNthTurnFromEnd(messages, keepToolTurns);
|
|
224
|
+
if (cutIndex <= 0) return messages.map(message => ({ ...message }));
|
|
225
|
+
|
|
226
|
+
const older = messages.slice(0, cutIndex);
|
|
227
|
+
const recent = messages.slice(cutIndex);
|
|
228
|
+
const cleanedOlder = [];
|
|
229
|
+
for (const message of older) {
|
|
230
|
+
if (!message || typeof message !== 'object') continue;
|
|
231
|
+
if (message.role === 'tool') continue;
|
|
232
|
+
|
|
233
|
+
const next = { ...message };
|
|
234
|
+
if (Array.isArray(next.toolCalls)) delete next.toolCalls;
|
|
235
|
+
if (Array.isArray(next.content)) next.content = stripToolContentParts(next.content);
|
|
236
|
+
if (next.role === 'assistant' && !hasContentAfterToolStrip(next.content)) continue;
|
|
237
|
+
if (next.role === 'user' && Array.isArray(next.content) && next.content.length === 0) continue;
|
|
238
|
+
cleanedOlder.push(next);
|
|
239
|
+
}
|
|
240
|
+
return [...cleanedOlder, ...recent.map(message => ({ ...message }))];
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function truncateTextToTokens(text, tokenBudget) {
|
|
244
|
+
if (typeof text !== 'string' || tokenBudget <= 0) return '';
|
|
245
|
+
if (estimateTokens(text) <= tokenBudget) return text;
|
|
246
|
+
let out = text.slice(0, Math.max(0, Math.floor(tokenBudget * TEXT_CHARS_PER_TOKEN)));
|
|
247
|
+
while (out && estimateTokens(out) > tokenBudget) out = out.slice(0, -1);
|
|
248
|
+
return out;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
function attachmentMarkerPart(remainingTokens) {
|
|
252
|
+
if (remainingTokens < estimateTokens(OVERSIZED_ATTACHMENT_MARKER)) return null;
|
|
253
|
+
return { type: 'text', text: OVERSIZED_ATTACHMENT_MARKER };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
function fitContentToBudget(content, tokenBudget) {
|
|
257
|
+
if (tokenBudget <= 0) return typeof content === 'string' ? '' : [];
|
|
258
|
+
if (typeof content === 'string') return truncateTextToTokens(content, tokenBudget);
|
|
259
|
+
if (!Array.isArray(content)) {
|
|
260
|
+
if (content && typeof content === 'object') {
|
|
261
|
+
const serialized = serializeJsonValue(content);
|
|
262
|
+
return estimateTokens(serialized) <= tokenBudget
|
|
263
|
+
? content
|
|
264
|
+
: truncateTextToTokens(serialized, tokenBudget);
|
|
265
|
+
}
|
|
266
|
+
return content;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
let remaining = Math.max(0, tokenBudget - CONTENT_PART_FRAME_TOKENS);
|
|
270
|
+
const out = [];
|
|
271
|
+
for (const part of content) {
|
|
272
|
+
const cost = estimateContentPartTokens(part);
|
|
273
|
+
if (typeof part === 'string') {
|
|
274
|
+
const text = truncateTextToTokens(part, remaining);
|
|
275
|
+
if (text) {
|
|
276
|
+
out.push(text);
|
|
277
|
+
remaining -= estimateTokens(text);
|
|
278
|
+
}
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
if (!part || typeof part !== 'object') {
|
|
282
|
+
if (cost <= remaining) {
|
|
283
|
+
out.push(part);
|
|
284
|
+
remaining -= cost;
|
|
285
|
+
}
|
|
286
|
+
continue;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
const type = String(part.type || '');
|
|
290
|
+
if (type === 'text' || type === 'input_text' || type === 'output_text') {
|
|
291
|
+
const text = truncateTextToTokens(typeof part.text === 'string' ? part.text : '', remaining);
|
|
292
|
+
if (text) {
|
|
293
|
+
out.push({ ...part, text });
|
|
294
|
+
remaining -= estimateTokens(text);
|
|
295
|
+
}
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
if (cost <= remaining) {
|
|
300
|
+
out.push({ ...part });
|
|
301
|
+
remaining -= cost;
|
|
302
|
+
continue;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// A non-binary structured part may still contain useful text/content.
|
|
306
|
+
// Let the generic text marker path below handle it only when it is
|
|
307
|
+
// genuinely not safely sliceable.
|
|
308
|
+
if (type === 'tool_result' || type === 'function_call_output') {
|
|
309
|
+
const marker = attachmentMarkerPart(remaining);
|
|
310
|
+
if (marker) {
|
|
311
|
+
out.push(marker);
|
|
312
|
+
remaining -= estimateTokens(marker.text);
|
|
313
|
+
}
|
|
314
|
+
continue;
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
// A binary part cannot be safely sliced. Replace it with a valid text
|
|
318
|
+
// marker rather than forwarding an oversized/invalid base64 payload.
|
|
319
|
+
const marker = attachmentMarkerPart(remaining);
|
|
320
|
+
if (marker) {
|
|
321
|
+
out.push(marker);
|
|
322
|
+
remaining -= estimateTokens(marker.text);
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
return out;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
function messageOverheadTokens(message) {
|
|
329
|
+
return estimateMessageTokens({ ...message, content: '' });
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
function hasProviderContent(content) {
|
|
333
|
+
if (typeof content === 'string') return content.trim().length > 0;
|
|
334
|
+
if (Array.isArray(content)) {
|
|
335
|
+
return content.some(part => {
|
|
336
|
+
if (typeof part === 'string') return part.trim().length > 0;
|
|
337
|
+
if (!part || typeof part !== 'object') return part != null;
|
|
338
|
+
if (typeof part.text === 'string') return part.text.trim().length > 0;
|
|
339
|
+
return part.type !== 'tool_use' && part.type !== 'tool_result'
|
|
340
|
+
&& part.type !== 'function_call' && part.type !== 'function_call_output';
|
|
341
|
+
});
|
|
342
|
+
}
|
|
343
|
+
return content != null;
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
function dropEmptyAssistantRows(messages) {
|
|
347
|
+
if (!Array.isArray(messages)) return [];
|
|
348
|
+
return messages.filter(message => {
|
|
349
|
+
if (!message || message.role !== 'assistant') return true;
|
|
350
|
+
return hasProviderContent(message.content)
|
|
351
|
+
|| (Array.isArray(message.toolCalls) && message.toolCalls.length > 0)
|
|
352
|
+
|| (Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0);
|
|
353
|
+
});
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
function shrinkMessageToBudget(message, tokenBudget) {
|
|
357
|
+
if (!message || typeof message !== 'object') return message;
|
|
358
|
+
const next = { ...message };
|
|
359
|
+
const hadThinkingBlocks = Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0;
|
|
360
|
+
const originalThinkingBlocks = validThinkingBlocks(message.thinkingBlocks);
|
|
361
|
+
const messageWithoutThinking = { ...message };
|
|
362
|
+
delete messageWithoutThinking.thinkingBlocks;
|
|
363
|
+
const contentBudget = Math.max(0, tokenBudget - messageOverheadTokens(messageWithoutThinking));
|
|
364
|
+
next.content = fitContentToBudget(message.content, contentBudget);
|
|
365
|
+
|
|
366
|
+
// Anthropic signed thinking blocks are atomic. Never truncate their payload
|
|
367
|
+
// or signature. Keep the complete block set only if it fits; otherwise omit
|
|
368
|
+
// the private replay state from this provider copy. Historical text/tool
|
|
369
|
+
// context remains usable, and the durable transcript remains untouched.
|
|
370
|
+
let thinkingBlocksKept = false;
|
|
371
|
+
if (hadThinkingBlocks && originalThinkingBlocks.length === 0) {
|
|
372
|
+
delete next.thinkingBlocks;
|
|
373
|
+
} else if (estimateMessageTokens({ ...next, thinkingBlocks: originalThinkingBlocks }) <= tokenBudget) {
|
|
374
|
+
if (originalThinkingBlocks.length > 0) {
|
|
375
|
+
next.thinkingBlocks = originalThinkingBlocks;
|
|
376
|
+
thinkingBlocksKept = true;
|
|
377
|
+
} else {
|
|
378
|
+
delete next.thinkingBlocks;
|
|
379
|
+
}
|
|
380
|
+
} else {
|
|
381
|
+
delete next.thinkingBlocks;
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
// Anthropic requires the signed thinking blocks that precede a tool_use
|
|
385
|
+
// within the same assistant turn. If the atomic thinking replay cannot fit,
|
|
386
|
+
// drop the complete tool arc from this provider copy; pairSanitize removes
|
|
387
|
+
// its role:'tool' rows below. Keeping toolCalls without their signed prefix
|
|
388
|
+
// would produce a protocol-invalid request.
|
|
389
|
+
if (hadThinkingBlocks && !thinkingBlocksKept && Array.isArray(next.toolCalls)) {
|
|
390
|
+
delete next.toolCalls;
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
// If tool metadata alone exceeds the allowance, remove the tool calls from
|
|
394
|
+
// this provider copy. pairSanitize will remove any now-orphaned tool rows;
|
|
395
|
+
// the durable transcript remains untouched.
|
|
396
|
+
if (estimateMessageTokens(next) > tokenBudget && Array.isArray(next.toolCalls)) {
|
|
397
|
+
delete next.toolCalls;
|
|
398
|
+
}
|
|
399
|
+
return next;
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
function dropOldestHistoryUntilBudget(messages, tokenBudget) {
|
|
403
|
+
let out = pairSanitize(messages);
|
|
404
|
+
let turns = countTurns(out);
|
|
405
|
+
while (estimateMessagesTokens(out) > tokenBudget && out.length > 0 && turns > 1) {
|
|
406
|
+
const next = pairSanitize(sliceLastNTurns(out, turns - 1));
|
|
407
|
+
if (next.length === out.length) break;
|
|
408
|
+
out = next;
|
|
409
|
+
turns = countTurns(out);
|
|
410
|
+
}
|
|
411
|
+
return out;
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
function providerUnits(messages) {
|
|
415
|
+
const units = [];
|
|
416
|
+
for (let index = 0; index < messages.length;) {
|
|
417
|
+
const message = messages[index];
|
|
418
|
+
if (message?.role !== 'assistant' || !Array.isArray(message.toolCalls) || message.toolCalls.length === 0) {
|
|
419
|
+
units.push([message]);
|
|
420
|
+
index += 1;
|
|
421
|
+
continue;
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
const callIds = new Set(message.toolCalls.map(call => call?.id).filter(Boolean));
|
|
425
|
+
const unit = [message];
|
|
426
|
+
let nextIndex = index + 1;
|
|
427
|
+
while (nextIndex < messages.length && messages[nextIndex]?.role === 'tool') {
|
|
428
|
+
const toolMessage = messages[nextIndex];
|
|
429
|
+
if (callIds.has(toolMessage.toolCallId)) unit.push(toolMessage);
|
|
430
|
+
nextIndex += 1;
|
|
431
|
+
}
|
|
432
|
+
units.push(unit);
|
|
433
|
+
index = nextIndex;
|
|
434
|
+
}
|
|
435
|
+
return units;
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
function fitProviderUnit(unit, tokenBudget) {
|
|
439
|
+
if (!Array.isArray(unit) || unit.length === 0 || tokenBudget <= 0) return [];
|
|
440
|
+
const [owner, ...toolMessages] = unit;
|
|
441
|
+
const isToolUnit = owner?.role === 'assistant'
|
|
442
|
+
&& Array.isArray(owner.toolCalls)
|
|
443
|
+
&& owner.toolCalls.length > 0
|
|
444
|
+
&& toolMessages.length > 0;
|
|
445
|
+
if (!isToolUnit) {
|
|
446
|
+
const fitted = shrinkMessageToBudget(owner, tokenBudget);
|
|
447
|
+
return dropEmptyAssistantRows([fitted]);
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
// Fit the assistant owner first. Signed thinking blocks are atomic; when
|
|
451
|
+
// they cannot fit, shrinkMessageToBudget removes the toolCalls as well, and
|
|
452
|
+
// this whole unit is dropped so no tool_result can become orphaned.
|
|
453
|
+
const fittedOwner = shrinkMessageToBudget(owner, tokenBudget);
|
|
454
|
+
if (!Array.isArray(fittedOwner.toolCalls) || fittedOwner.toolCalls.length === 0) return [];
|
|
455
|
+
|
|
456
|
+
const fitted = [fittedOwner];
|
|
457
|
+
let remaining = Math.max(0, tokenBudget - estimateMessageTokens(fittedOwner));
|
|
458
|
+
for (const toolMessage of toolMessages) {
|
|
459
|
+
if (messageOverheadTokens(toolMessage) > remaining) return [];
|
|
460
|
+
const fittedTool = shrinkMessageToBudget(toolMessage, remaining);
|
|
461
|
+
const fittedToolTokens = estimateMessageTokens(fittedTool);
|
|
462
|
+
if (fittedToolTokens > remaining) return [];
|
|
463
|
+
fitted.push(fittedTool);
|
|
464
|
+
remaining -= fittedToolTokens;
|
|
465
|
+
}
|
|
466
|
+
return fitted;
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
function fitMessagesToBudget(messages, tokenBudget) {
|
|
470
|
+
let out = dropOldestHistoryUntilBudget(pairSanitize(messages), tokenBudget);
|
|
471
|
+
if (estimateMessagesTokens(out) <= tokenBudget) return dropEmptyAssistantRows(out);
|
|
472
|
+
|
|
473
|
+
// Treat assistant(toolCalls)+tool rows as one provider unit. The newest unit
|
|
474
|
+
// gets the remaining budget first, but its paired tool results share that
|
|
475
|
+
// budget with the assistant owner. This preserves valid tool protocol shape
|
|
476
|
+
// while bounding serialized object output and signed thinking together.
|
|
477
|
+
const units = providerUnits(out);
|
|
478
|
+
const fittedUnits = Array.from({ length: units.length }, () => []);
|
|
479
|
+
let reserved = 0;
|
|
480
|
+
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
481
|
+
const fitted = fitProviderUnit(units[index], Math.max(0, tokenBudget - reserved));
|
|
482
|
+
fittedUnits[index] = fitted;
|
|
483
|
+
reserved += estimateMessagesTokens(fitted);
|
|
484
|
+
}
|
|
485
|
+
out = fittedUnits.flat();
|
|
486
|
+
return dropEmptyAssistantRows(pairSanitize(out));
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
function truncateToolResultsForModel(messages, options = {}) {
|
|
490
|
+
if (!Array.isArray(messages) || messages.length === 0) return [];
|
|
491
|
+
return messages.map(message => {
|
|
492
|
+
if (!message || message.role !== 'tool' || typeof message.content !== 'string') {
|
|
493
|
+
return { ...message };
|
|
494
|
+
}
|
|
495
|
+
return {
|
|
496
|
+
...message,
|
|
497
|
+
content: truncateToolResultIfNeeded(message.content, {
|
|
498
|
+
toolName: message.name || message.toolName || 'tool_result',
|
|
499
|
+
language: options.language,
|
|
500
|
+
}),
|
|
501
|
+
};
|
|
502
|
+
});
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
/**
|
|
506
|
+
* Build a bounded, pair-safe copy for one provider request.
|
|
507
|
+
*
|
|
508
|
+
* The transform is deterministic and non-persistent:
|
|
509
|
+
* 1. keep at most `recentTurnCap` turns and `maxMessageCount` rows;
|
|
510
|
+
* 2. drop oldest turns until the configured approximate message budget fits;
|
|
511
|
+
* 3. remove old tool noise;
|
|
512
|
+
* 4. bound large tool-result bodies and multimodal content;
|
|
513
|
+
* 5. remove orphan tool pairs.
|
|
514
|
+
*
|
|
515
|
+
* @param {Array<object>} snapshot
|
|
516
|
+
* @param {{ messageTokenBudget?: number, recentTurnCap?: number, maxMessageCount?: number, keepToolTurns?: number, language?: string }} [options]
|
|
517
|
+
* @returns {Array<object>}
|
|
518
|
+
*/
|
|
519
|
+
export function trimSnapshotForBudget(snapshot, options = {}) {
|
|
520
|
+
if (!Array.isArray(snapshot) || snapshot.length === 0) return [];
|
|
521
|
+
|
|
522
|
+
const recentTurnCap = Number.isFinite(options.recentTurnCap) && options.recentTurnCap > 0
|
|
523
|
+
? Math.floor(options.recentTurnCap)
|
|
524
|
+
: DEFAULT_RECENT_TURN_CAP;
|
|
525
|
+
const messageTokenBudget = Number.isFinite(options.messageTokenBudget) && options.messageTokenBudget > 0
|
|
526
|
+
? Math.floor(options.messageTokenBudget)
|
|
527
|
+
: DEFAULT_MESSAGE_TOKEN_BUDGET;
|
|
528
|
+
const maxMessageCount = Number.isFinite(options.maxMessageCount) && options.maxMessageCount > 0
|
|
529
|
+
? Math.floor(options.maxMessageCount)
|
|
530
|
+
: DEFAULT_RUNTIME_CACHE_MESSAGE_CAP;
|
|
531
|
+
|
|
532
|
+
let trimmed = sliceLastNTurns(snapshot, recentTurnCap);
|
|
533
|
+
if (trimmed.length > maxMessageCount) trimmed = trimmed.slice(-maxMessageCount);
|
|
534
|
+
let remainingTurnCap = recentTurnCap;
|
|
535
|
+
let tokens = estimateMessagesTokens(trimmed);
|
|
536
|
+
while (tokens > messageTokenBudget && remainingTurnCap > 1) {
|
|
537
|
+
const nextTurnCap = remainingTurnCap - 1;
|
|
538
|
+
const next = sliceLastNTurns(trimmed, nextTurnCap);
|
|
539
|
+
if (next.length === trimmed.length) break;
|
|
540
|
+
remainingTurnCap = nextTurnCap;
|
|
541
|
+
trimmed = next;
|
|
542
|
+
tokens = estimateMessagesTokens(trimmed);
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
trimmed = stripToolNoiseFromOlderTurns(trimmed, {
|
|
546
|
+
keepToolTurns: options.keepToolTurns,
|
|
547
|
+
});
|
|
548
|
+
trimmed = truncateToolResultsForModel(trimmed, { language: options.language });
|
|
549
|
+
trimmed = pairSanitize(trimmed);
|
|
550
|
+
return fitMessagesToBudget(trimmed, messageTokenBudget);
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
/**
|
|
554
|
+
* Bound the Session-level runtime history cache. This is deliberately stricter
|
|
555
|
+
* than the provider configuration: the cache is only a disposable source
|
|
556
|
+
* snapshot, while ConversationStore retains the complete transcript.
|
|
557
|
+
*
|
|
558
|
+
* @param {Array<object>} snapshot
|
|
559
|
+
* @param {{ language?: string }} [options]
|
|
560
|
+
* @returns {Array<object>}
|
|
561
|
+
*/
|
|
562
|
+
export function trimHistoryCacheForRuntime(snapshot, options = {}) {
|
|
563
|
+
return trimSnapshotForBudget(snapshot, {
|
|
564
|
+
recentTurnCap: DEFAULT_RUNTIME_CACHE_TURN_CAP,
|
|
565
|
+
messageTokenBudget: DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET,
|
|
566
|
+
maxMessageCount: DEFAULT_RUNTIME_CACHE_MESSAGE_CAP,
|
|
567
|
+
keepToolTurns: DEFAULT_KEEP_TOOL_TURNS,
|
|
568
|
+
language: options.language,
|
|
569
|
+
});
|
|
570
|
+
}
|
package/yeaft/llm/adapter.js
CHANGED
|
@@ -163,7 +163,7 @@ export function classifyPolicyError(statusCode, responseBody = '', details = {})
|
|
|
163
163
|
return new LLMPolicyError(signals.message, status, details);
|
|
164
164
|
}
|
|
165
165
|
|
|
166
|
-
/** Context too long error (413 or API-specific)
|
|
166
|
+
/** Context too long error (413 or API-specific). */
|
|
167
167
|
export class LLMContextError extends Error {
|
|
168
168
|
constructor(message) {
|
|
169
169
|
super(message);
|
package/yeaft/llm/anthropic.js
CHANGED
|
@@ -534,7 +534,7 @@ export class AnthropicAdapter extends LLMAdapter {
|
|
|
534
534
|
* Non-streaming call for side queries.
|
|
535
535
|
*
|
|
536
536
|
* task-327c: accepts `effort` for internal scenario-tagged calls
|
|
537
|
-
* (
|
|
537
|
+
* (dream/recall/light). Guards mirror stream() — unsupported
|
|
538
538
|
* models silently drop the param. max_tokens auto-widens to budget+1024
|
|
539
539
|
* when needed.
|
|
540
540
|
*/
|
package/yeaft/llm/models-dev.js
CHANGED
|
@@ -184,7 +184,8 @@ export async function listProviderModels(providerId, { yeaftDir = null } = {}) {
|
|
|
184
184
|
* that provider's numbers verbatim — the caller knew which gateway it
|
|
185
185
|
* was talking to.
|
|
186
186
|
* • Otherwise we take the MIN of every provider's `context` and `output`.
|
|
187
|
-
* Context is a ceiling: under-shooting risks an
|
|
187
|
+
* Context is a ceiling: under-shooting risks an unnecessarily small
|
|
188
|
+
* request window; over-shooting risks an LLMContextError mid-query (worse);
|
|
188
189
|
* over-shooting risks an LLMContextError mid-query (worse). Min picks
|
|
189
190
|
* the safer side. Users who know better can pin numbers explicitly via
|
|
190
191
|
* `providers[].models[].contextWindow` in `~/.yeaft/config.json`.
|
|
@@ -521,7 +521,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
|
|
|
521
521
|
// ─── Non-streaming call() ───────────────────────────────
|
|
522
522
|
|
|
523
523
|
/**
|
|
524
|
-
* Side-query (
|
|
524
|
+
* Side-query (dream / recall / light) entry point. Does
|
|
525
525
|
* NOT accept `onRawExchange` — these calls intentionally don't surface
|
|
526
526
|
* in the user-facing debug panel. If a future product change wants to
|
|
527
527
|
* expose them, mirror the stream() instrumentation. Parity with
|
package/yeaft/llm/router.js
CHANGED
|
@@ -147,9 +147,8 @@ export function filterEffortForModel(params, context = {}) {
|
|
|
147
147
|
/**
|
|
148
148
|
* task-715: last-line-of-defense pair sanitize at the wire.
|
|
149
149
|
*
|
|
150
|
-
* `pairSanitize` already runs in
|
|
151
|
-
* (`conversation/persist.js#loadRecentBySession`
|
|
152
|
-
* `history-compact.js#compactHistory`), but the engine's main loop
|
|
150
|
+
* `pairSanitize` already runs in the persisted-history path
|
|
151
|
+
* (`conversation/persist.js#loadRecentBySession`), but the engine's main loop
|
|
153
152
|
* mutates `conversationMessages` AFTER those — appending tool results
|
|
154
153
|
* mid-loop, archiving bulky tool results into stubs, and (in failure
|
|
155
154
|
* paths) potentially leaving an assistant `tool_use` whose matching
|
|
@@ -48,8 +48,8 @@ function addUsage(total, usage) {
|
|
|
48
48
|
* reports once after its event stream finishes or aborts, and every
|
|
49
49
|
* non-streaming side call reports once whether it succeeds or fails.
|
|
50
50
|
*
|
|
51
|
-
* Parent VP engines, sub-agent engines, Dream,
|
|
52
|
-
*
|
|
51
|
+
* Parent VP engines, sub-agent engines, Dream, reflection, AMS, and classifiers
|
|
52
|
+
* all reuse this adapter, so none need their own accounting hook.
|
|
53
53
|
*/
|
|
54
54
|
export class UsageAccountingAdapter extends LLMAdapter {
|
|
55
55
|
#adapter;
|
package/yeaft/pair-sanitize.js
CHANGED
|
@@ -3,9 +3,9 @@
|
|
|
3
3
|
* slice so it can be safely fed to the LLM adapter.
|
|
4
4
|
*
|
|
5
5
|
* Why this exists:
|
|
6
|
-
* `agent/yeaft/conversation/persist.js#loadRecentBySession` and
|
|
7
|
-
*
|
|
8
|
-
*
|
|
6
|
+
* `agent/yeaft/conversation/persist.js#loadRecentBySession` and the
|
|
7
|
+
* deterministic provider history window both produce sub-slices of a longer
|
|
8
|
+
* message stream. Both paths can — depending on
|
|
9
9
|
* where the cut lands — produce one of two illegal shapes:
|
|
10
10
|
* 1. A `role: 'tool'` message whose owning assistant `tool_use` is
|
|
11
11
|
* no longer in the slice.
|
package/yeaft/prompts.js
CHANGED
|
@@ -16,10 +16,9 @@
|
|
|
16
16
|
* ④ Active Scope — structured per-turn scope summary
|
|
17
17
|
* (session / vp / members / envelope IDs)
|
|
18
18
|
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
* AMS Resident.
|
|
19
|
+
* Long-term semantic context comes only from the AMS Memory outlet. The
|
|
20
|
+
* conversation transcript stays in the messages timeline and is bounded by
|
|
21
|
+
* deterministic per-request history-window trimming.
|
|
23
22
|
*/
|
|
24
23
|
|
|
25
24
|
import { readFileSync, existsSync } from 'fs';
|