@yeaft/webchat-agent 1.0.447 → 1.0.448

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,570 @@
1
+ /**
2
+ * history-window.js — deterministic history shaping for provider requests.
3
+ *
4
+ * This module never calls an LLM, writes a summary, archives transcript rows,
5
+ * or changes the persisted conversation. It builds bounded copies; callers may
6
+ * replace a disposable runtime cache with one of those copies. The persisted
7
+ * message history remains authoritative; Memory/Dream is the long-lived
8
+ * semantic context.
9
+ */
10
+
11
+ import { estimateTokens } from './conversation/persist.js';
12
+ import { pairSanitize } from './pair-sanitize.js';
13
+ import { truncateToolResultIfNeeded } from './tools/registry.js';
14
+ import { countTurns, indexOfNthTurnFromEnd, sliceLastNTurns } from './turn-utils.js';
15
+
16
+ export const DEFAULT_KEEP_TOOL_TURNS = 3;
17
+ export const DEFAULT_RECENT_TURN_CAP = 25;
18
+ export const DEFAULT_MESSAGE_TOKEN_BUDGET = 32768;
19
+
20
+ // Runtime history is a cache, not a second transcript. Keep its hard cap
21
+ // independent from a user-configured provider budget so a large config cannot
22
+ // turn the bridge cache back into an unbounded transcript.
23
+ export const DEFAULT_RUNTIME_CACHE_TURN_CAP = 25;
24
+ export const DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET = 32768;
25
+ export const DEFAULT_RUNTIME_CACHE_MESSAGE_CAP = 256;
26
+
27
+ const IMAGE_PART_TOKEN_COST = 1024;
28
+ const DOCUMENT_PART_TOKEN_COST = 2048;
29
+ const CONTENT_PART_FRAME_TOKENS = 2;
30
+ const TEXT_CHARS_PER_TOKEN = 4;
31
+ const BINARY_CHARS_PER_TOKEN = 16;
32
+ const OVERSIZED_ATTACHMENT_MARKER = '[attachment omitted from provider context budget]';
33
+
34
+ function serializeJsonValue(value) {
35
+ if (typeof value === 'string') return value;
36
+ try {
37
+ const serialized = JSON.stringify(value ?? '');
38
+ return typeof serialized === 'string' ? serialized : String(value ?? '');
39
+ } catch {
40
+ return String(value ?? '');
41
+ }
42
+ }
43
+
44
+ function safeJsonTokenEstimate(value) {
45
+ return estimateTokens(serializeJsonValue(value));
46
+ }
47
+
48
+ function thinkingBlockWirePart(block) {
49
+ if (!block || typeof block !== 'object') return null;
50
+ if (typeof block.signature !== 'string' || !block.signature) return null;
51
+ if (block.redacted) {
52
+ if (typeof block.data !== 'string') return null;
53
+ return { type: 'redacted_thinking', data: block.data, signature: block.signature };
54
+ }
55
+ if (typeof block.thinking !== 'string') return null;
56
+ return { type: 'thinking', thinking: block.thinking, signature: block.signature };
57
+ }
58
+
59
+ function validThinkingBlocks(blocks) {
60
+ if (!Array.isArray(blocks)) return [];
61
+ return blocks.filter(block => thinkingBlockWirePart(block));
62
+ }
63
+
64
+ function estimateThinkingBlockTokens(block) {
65
+ const part = thinkingBlockWirePart(block);
66
+ return part ? estimateContentPartTokens(part) : safeJsonTokenEstimate(block);
67
+ }
68
+
69
+ function estimateThinkingBlocksTokens(blocks) {
70
+ if (!Array.isArray(blocks) || blocks.length === 0) return 0;
71
+ return blocks.reduce((total, block) => total + estimateThinkingBlockTokens(block), CONTENT_PART_FRAME_TOKENS);
72
+ }
73
+
74
+ function binaryPayloadTokenEstimate(value) {
75
+ if (typeof value !== 'string' || value.length === 0) return 0;
76
+ // Base64/image bytes are not text tokens, so do not charge them at the text
77
+ // ratio. Still count a conservative wire/configuration cost; otherwise a
78
+ // huge content part would bypass the request budget entirely.
79
+ return Math.ceil(value.length / BINARY_CHARS_PER_TOKEN);
80
+ }
81
+
82
+ function partMetadataTokenEstimate(part, fields = []) {
83
+ return fields.reduce((total, field) => {
84
+ const value = part?.[field];
85
+ return total + (typeof value === 'string' ? estimateTokens(value) : 0);
86
+ }, 0);
87
+ }
88
+
89
+ /**
90
+ * Estimate one provider content part. This is a guardrail, not a tokenizer;
91
+ * the provider remains authoritative about the actual context limit.
92
+ *
93
+ * @param {unknown} part
94
+ * @returns {number}
95
+ */
96
+ export function estimateContentPartTokens(part) {
97
+ if (typeof part === 'string') return estimateTokens(part);
98
+ if (!part || typeof part !== 'object') return estimateTokens(String(part ?? ''));
99
+
100
+ const type = String(part.type || '');
101
+ if (type === 'text' || type === 'input_text' || type === 'output_text') {
102
+ return estimateTokens(typeof part.text === 'string' ? part.text : '');
103
+ }
104
+ if (type === 'thinking') {
105
+ return 4 + estimateTokens(part.thinking || '') + estimateTokens(part.signature || '');
106
+ }
107
+ if (type === 'redacted_thinking') {
108
+ return 4 + estimateTokens(part.data || '') + estimateTokens(part.signature || '');
109
+ }
110
+ if (type === 'image' || type === 'input_image') {
111
+ const source = part.source && typeof part.source === 'object' ? part.source : part;
112
+ return IMAGE_PART_TOKEN_COST
113
+ + partMetadataTokenEstimate(part, ['title', 'alt', 'image_url'])
114
+ + partMetadataTokenEstimate(source, ['url', 'media_type', 'mediaType'])
115
+ + binaryPayloadTokenEstimate(source.data);
116
+ }
117
+ if (type === 'document' || type === 'input_file') {
118
+ const source = part.source && typeof part.source === 'object' ? part.source : part;
119
+ return DOCUMENT_PART_TOKEN_COST
120
+ + partMetadataTokenEstimate(part, ['title', 'filename', 'file_data'])
121
+ + partMetadataTokenEstimate(source, ['media_type', 'mediaType', 'url'])
122
+ + binaryPayloadTokenEstimate(source.data);
123
+ }
124
+ if (type === 'tool_result') {
125
+ return 4 + estimateContentTokens(part.content);
126
+ }
127
+ if (type === 'function_call_output') {
128
+ return 4 + estimateTokens(serializeJsonValue(part.output));
129
+ }
130
+ return safeJsonTokenEstimate(part);
131
+ }
132
+
133
+ /**
134
+ * Estimate string or array provider content, including multimodal parts.
135
+ *
136
+ * @param {unknown} content
137
+ * @returns {number}
138
+ */
139
+ export function estimateContentTokens(content) {
140
+ if (typeof content === 'string') return estimateTokens(content);
141
+ if (Array.isArray(content)) {
142
+ return CONTENT_PART_FRAME_TOKENS
143
+ + content.reduce((total, part) => total + estimateContentPartTokens(part), 0);
144
+ }
145
+ if (content == null) return 0;
146
+ return safeJsonTokenEstimate(content);
147
+ }
148
+
149
+ /**
150
+ * Estimate the provider-token weight of one message.
151
+ *
152
+ * @param {object} message
153
+ * @returns {number}
154
+ */
155
+ export function estimateMessageTokens(message) {
156
+ if (!message || typeof message !== 'object') return 0;
157
+ let total = 2 + estimateContentTokens(message.content);
158
+ total += estimateThinkingBlocksTokens(message.thinkingBlocks);
159
+ if (Array.isArray(message.toolCalls)) {
160
+ for (const toolCall of message.toolCalls) {
161
+ total += 4;
162
+ try {
163
+ const input = typeof toolCall.input === 'string'
164
+ ? toolCall.input
165
+ : JSON.stringify(toolCall.input || {});
166
+ total += estimateTokens(input);
167
+ } catch {
168
+ // Ignore malformed tool input in the approximate guardrail.
169
+ }
170
+ if (toolCall.name) total += estimateTokens(toolCall.name);
171
+ }
172
+ }
173
+ if (message.toolCallId) total += 2;
174
+ return total;
175
+ }
176
+
177
+ /**
178
+ * @param {Array<object>} messages
179
+ * @returns {number}
180
+ */
181
+ export function estimateMessagesTokens(messages) {
182
+ if (!Array.isArray(messages)) return 0;
183
+ return messages.reduce((total, message) => total + estimateMessageTokens(message), 0);
184
+ }
185
+
186
+ function hasContentAfterToolStrip(content) {
187
+ if (typeof content === 'string') return content.trim().length > 0;
188
+ if (Array.isArray(content)) {
189
+ return content.some(part => {
190
+ if (typeof part === 'string') return part.trim().length > 0;
191
+ if (!part || typeof part !== 'object') return part != null;
192
+ return typeof part.text === 'string' ? part.text.trim().length > 0 : true;
193
+ });
194
+ }
195
+ return content != null;
196
+ }
197
+
198
+ function stripToolContentParts(content) {
199
+ if (!Array.isArray(content)) return content;
200
+ return content.filter(part => {
201
+ if (!part || typeof part !== 'object') return true;
202
+ return part.type !== 'tool_use'
203
+ && part.type !== 'tool_result'
204
+ && part.type !== 'function_call'
205
+ && part.type !== 'function_call_output';
206
+ });
207
+ }
208
+
209
+ /**
210
+ * Remove old tool payloads from the provider copy while keeping ordinary user
211
+ * and assistant text. The newest tool turns remain lossless so the active tool
212
+ * protocol stays paired; pairSanitize runs after this transform.
213
+ *
214
+ * @param {Array<object>} messages
215
+ * @param {{ keepToolTurns?: number }} [options]
216
+ * @returns {Array<object>}
217
+ */
218
+ export function stripToolNoiseFromOlderTurns(messages, options = {}) {
219
+ if (!Array.isArray(messages) || messages.length === 0) return [];
220
+ const keepToolTurns = Number.isFinite(options.keepToolTurns) && options.keepToolTurns >= 0
221
+ ? options.keepToolTurns
222
+ : DEFAULT_KEEP_TOOL_TURNS;
223
+ const cutIndex = indexOfNthTurnFromEnd(messages, keepToolTurns);
224
+ if (cutIndex <= 0) return messages.map(message => ({ ...message }));
225
+
226
+ const older = messages.slice(0, cutIndex);
227
+ const recent = messages.slice(cutIndex);
228
+ const cleanedOlder = [];
229
+ for (const message of older) {
230
+ if (!message || typeof message !== 'object') continue;
231
+ if (message.role === 'tool') continue;
232
+
233
+ const next = { ...message };
234
+ if (Array.isArray(next.toolCalls)) delete next.toolCalls;
235
+ if (Array.isArray(next.content)) next.content = stripToolContentParts(next.content);
236
+ if (next.role === 'assistant' && !hasContentAfterToolStrip(next.content)) continue;
237
+ if (next.role === 'user' && Array.isArray(next.content) && next.content.length === 0) continue;
238
+ cleanedOlder.push(next);
239
+ }
240
+ return [...cleanedOlder, ...recent.map(message => ({ ...message }))];
241
+ }
242
+
243
+ function truncateTextToTokens(text, tokenBudget) {
244
+ if (typeof text !== 'string' || tokenBudget <= 0) return '';
245
+ if (estimateTokens(text) <= tokenBudget) return text;
246
+ let out = text.slice(0, Math.max(0, Math.floor(tokenBudget * TEXT_CHARS_PER_TOKEN)));
247
+ while (out && estimateTokens(out) > tokenBudget) out = out.slice(0, -1);
248
+ return out;
249
+ }
250
+
251
+ function attachmentMarkerPart(remainingTokens) {
252
+ if (remainingTokens < estimateTokens(OVERSIZED_ATTACHMENT_MARKER)) return null;
253
+ return { type: 'text', text: OVERSIZED_ATTACHMENT_MARKER };
254
+ }
255
+
256
+ function fitContentToBudget(content, tokenBudget) {
257
+ if (tokenBudget <= 0) return typeof content === 'string' ? '' : [];
258
+ if (typeof content === 'string') return truncateTextToTokens(content, tokenBudget);
259
+ if (!Array.isArray(content)) {
260
+ if (content && typeof content === 'object') {
261
+ const serialized = serializeJsonValue(content);
262
+ return estimateTokens(serialized) <= tokenBudget
263
+ ? content
264
+ : truncateTextToTokens(serialized, tokenBudget);
265
+ }
266
+ return content;
267
+ }
268
+
269
+ let remaining = Math.max(0, tokenBudget - CONTENT_PART_FRAME_TOKENS);
270
+ const out = [];
271
+ for (const part of content) {
272
+ const cost = estimateContentPartTokens(part);
273
+ if (typeof part === 'string') {
274
+ const text = truncateTextToTokens(part, remaining);
275
+ if (text) {
276
+ out.push(text);
277
+ remaining -= estimateTokens(text);
278
+ }
279
+ continue;
280
+ }
281
+ if (!part || typeof part !== 'object') {
282
+ if (cost <= remaining) {
283
+ out.push(part);
284
+ remaining -= cost;
285
+ }
286
+ continue;
287
+ }
288
+
289
+ const type = String(part.type || '');
290
+ if (type === 'text' || type === 'input_text' || type === 'output_text') {
291
+ const text = truncateTextToTokens(typeof part.text === 'string' ? part.text : '', remaining);
292
+ if (text) {
293
+ out.push({ ...part, text });
294
+ remaining -= estimateTokens(text);
295
+ }
296
+ continue;
297
+ }
298
+
299
+ if (cost <= remaining) {
300
+ out.push({ ...part });
301
+ remaining -= cost;
302
+ continue;
303
+ }
304
+
305
+ // A non-binary structured part may still contain useful text/content.
306
+ // Let the generic text marker path below handle it only when it is
307
+ // genuinely not safely sliceable.
308
+ if (type === 'tool_result' || type === 'function_call_output') {
309
+ const marker = attachmentMarkerPart(remaining);
310
+ if (marker) {
311
+ out.push(marker);
312
+ remaining -= estimateTokens(marker.text);
313
+ }
314
+ continue;
315
+ }
316
+
317
+ // A binary part cannot be safely sliced. Replace it with a valid text
318
+ // marker rather than forwarding an oversized/invalid base64 payload.
319
+ const marker = attachmentMarkerPart(remaining);
320
+ if (marker) {
321
+ out.push(marker);
322
+ remaining -= estimateTokens(marker.text);
323
+ }
324
+ }
325
+ return out;
326
+ }
327
+
328
+ function messageOverheadTokens(message) {
329
+ return estimateMessageTokens({ ...message, content: '' });
330
+ }
331
+
332
+ function hasProviderContent(content) {
333
+ if (typeof content === 'string') return content.trim().length > 0;
334
+ if (Array.isArray(content)) {
335
+ return content.some(part => {
336
+ if (typeof part === 'string') return part.trim().length > 0;
337
+ if (!part || typeof part !== 'object') return part != null;
338
+ if (typeof part.text === 'string') return part.text.trim().length > 0;
339
+ return part.type !== 'tool_use' && part.type !== 'tool_result'
340
+ && part.type !== 'function_call' && part.type !== 'function_call_output';
341
+ });
342
+ }
343
+ return content != null;
344
+ }
345
+
346
+ function dropEmptyAssistantRows(messages) {
347
+ if (!Array.isArray(messages)) return [];
348
+ return messages.filter(message => {
349
+ if (!message || message.role !== 'assistant') return true;
350
+ return hasProviderContent(message.content)
351
+ || (Array.isArray(message.toolCalls) && message.toolCalls.length > 0)
352
+ || (Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0);
353
+ });
354
+ }
355
+
356
+ function shrinkMessageToBudget(message, tokenBudget) {
357
+ if (!message || typeof message !== 'object') return message;
358
+ const next = { ...message };
359
+ const hadThinkingBlocks = Array.isArray(message.thinkingBlocks) && message.thinkingBlocks.length > 0;
360
+ const originalThinkingBlocks = validThinkingBlocks(message.thinkingBlocks);
361
+ const messageWithoutThinking = { ...message };
362
+ delete messageWithoutThinking.thinkingBlocks;
363
+ const contentBudget = Math.max(0, tokenBudget - messageOverheadTokens(messageWithoutThinking));
364
+ next.content = fitContentToBudget(message.content, contentBudget);
365
+
366
+ // Anthropic signed thinking blocks are atomic. Never truncate their payload
367
+ // or signature. Keep the complete block set only if it fits; otherwise omit
368
+ // the private replay state from this provider copy. Historical text/tool
369
+ // context remains usable, and the durable transcript remains untouched.
370
+ let thinkingBlocksKept = false;
371
+ if (hadThinkingBlocks && originalThinkingBlocks.length === 0) {
372
+ delete next.thinkingBlocks;
373
+ } else if (estimateMessageTokens({ ...next, thinkingBlocks: originalThinkingBlocks }) <= tokenBudget) {
374
+ if (originalThinkingBlocks.length > 0) {
375
+ next.thinkingBlocks = originalThinkingBlocks;
376
+ thinkingBlocksKept = true;
377
+ } else {
378
+ delete next.thinkingBlocks;
379
+ }
380
+ } else {
381
+ delete next.thinkingBlocks;
382
+ }
383
+
384
+ // Anthropic requires the signed thinking blocks that precede a tool_use
385
+ // within the same assistant turn. If the atomic thinking replay cannot fit,
386
+ // drop the complete tool arc from this provider copy; pairSanitize removes
387
+ // its role:'tool' rows below. Keeping toolCalls without their signed prefix
388
+ // would produce a protocol-invalid request.
389
+ if (hadThinkingBlocks && !thinkingBlocksKept && Array.isArray(next.toolCalls)) {
390
+ delete next.toolCalls;
391
+ }
392
+
393
+ // If tool metadata alone exceeds the allowance, remove the tool calls from
394
+ // this provider copy. pairSanitize will remove any now-orphaned tool rows;
395
+ // the durable transcript remains untouched.
396
+ if (estimateMessageTokens(next) > tokenBudget && Array.isArray(next.toolCalls)) {
397
+ delete next.toolCalls;
398
+ }
399
+ return next;
400
+ }
401
+
402
+ function dropOldestHistoryUntilBudget(messages, tokenBudget) {
403
+ let out = pairSanitize(messages);
404
+ let turns = countTurns(out);
405
+ while (estimateMessagesTokens(out) > tokenBudget && out.length > 0 && turns > 1) {
406
+ const next = pairSanitize(sliceLastNTurns(out, turns - 1));
407
+ if (next.length === out.length) break;
408
+ out = next;
409
+ turns = countTurns(out);
410
+ }
411
+ return out;
412
+ }
413
+
414
+ function providerUnits(messages) {
415
+ const units = [];
416
+ for (let index = 0; index < messages.length;) {
417
+ const message = messages[index];
418
+ if (message?.role !== 'assistant' || !Array.isArray(message.toolCalls) || message.toolCalls.length === 0) {
419
+ units.push([message]);
420
+ index += 1;
421
+ continue;
422
+ }
423
+
424
+ const callIds = new Set(message.toolCalls.map(call => call?.id).filter(Boolean));
425
+ const unit = [message];
426
+ let nextIndex = index + 1;
427
+ while (nextIndex < messages.length && messages[nextIndex]?.role === 'tool') {
428
+ const toolMessage = messages[nextIndex];
429
+ if (callIds.has(toolMessage.toolCallId)) unit.push(toolMessage);
430
+ nextIndex += 1;
431
+ }
432
+ units.push(unit);
433
+ index = nextIndex;
434
+ }
435
+ return units;
436
+ }
437
+
438
+ function fitProviderUnit(unit, tokenBudget) {
439
+ if (!Array.isArray(unit) || unit.length === 0 || tokenBudget <= 0) return [];
440
+ const [owner, ...toolMessages] = unit;
441
+ const isToolUnit = owner?.role === 'assistant'
442
+ && Array.isArray(owner.toolCalls)
443
+ && owner.toolCalls.length > 0
444
+ && toolMessages.length > 0;
445
+ if (!isToolUnit) {
446
+ const fitted = shrinkMessageToBudget(owner, tokenBudget);
447
+ return dropEmptyAssistantRows([fitted]);
448
+ }
449
+
450
+ // Fit the assistant owner first. Signed thinking blocks are atomic; when
451
+ // they cannot fit, shrinkMessageToBudget removes the toolCalls as well, and
452
+ // this whole unit is dropped so no tool_result can become orphaned.
453
+ const fittedOwner = shrinkMessageToBudget(owner, tokenBudget);
454
+ if (!Array.isArray(fittedOwner.toolCalls) || fittedOwner.toolCalls.length === 0) return [];
455
+
456
+ const fitted = [fittedOwner];
457
+ let remaining = Math.max(0, tokenBudget - estimateMessageTokens(fittedOwner));
458
+ for (const toolMessage of toolMessages) {
459
+ if (messageOverheadTokens(toolMessage) > remaining) return [];
460
+ const fittedTool = shrinkMessageToBudget(toolMessage, remaining);
461
+ const fittedToolTokens = estimateMessageTokens(fittedTool);
462
+ if (fittedToolTokens > remaining) return [];
463
+ fitted.push(fittedTool);
464
+ remaining -= fittedToolTokens;
465
+ }
466
+ return fitted;
467
+ }
468
+
469
+ function fitMessagesToBudget(messages, tokenBudget) {
470
+ let out = dropOldestHistoryUntilBudget(pairSanitize(messages), tokenBudget);
471
+ if (estimateMessagesTokens(out) <= tokenBudget) return dropEmptyAssistantRows(out);
472
+
473
+ // Treat assistant(toolCalls)+tool rows as one provider unit. The newest unit
474
+ // gets the remaining budget first, but its paired tool results share that
475
+ // budget with the assistant owner. This preserves valid tool protocol shape
476
+ // while bounding serialized object output and signed thinking together.
477
+ const units = providerUnits(out);
478
+ const fittedUnits = Array.from({ length: units.length }, () => []);
479
+ let reserved = 0;
480
+ for (let index = units.length - 1; index >= 0; index -= 1) {
481
+ const fitted = fitProviderUnit(units[index], Math.max(0, tokenBudget - reserved));
482
+ fittedUnits[index] = fitted;
483
+ reserved += estimateMessagesTokens(fitted);
484
+ }
485
+ out = fittedUnits.flat();
486
+ return dropEmptyAssistantRows(pairSanitize(out));
487
+ }
488
+
489
+ function truncateToolResultsForModel(messages, options = {}) {
490
+ if (!Array.isArray(messages) || messages.length === 0) return [];
491
+ return messages.map(message => {
492
+ if (!message || message.role !== 'tool' || typeof message.content !== 'string') {
493
+ return { ...message };
494
+ }
495
+ return {
496
+ ...message,
497
+ content: truncateToolResultIfNeeded(message.content, {
498
+ toolName: message.name || message.toolName || 'tool_result',
499
+ language: options.language,
500
+ }),
501
+ };
502
+ });
503
+ }
504
+
505
+ /**
506
+ * Build a bounded, pair-safe copy for one provider request.
507
+ *
508
+ * The transform is deterministic and non-persistent:
509
+ * 1. keep at most `recentTurnCap` turns and `maxMessageCount` rows;
510
+ * 2. drop oldest turns until the configured approximate message budget fits;
511
+ * 3. remove old tool noise;
512
+ * 4. bound large tool-result bodies and multimodal content;
513
+ * 5. remove orphan tool pairs.
514
+ *
515
+ * @param {Array<object>} snapshot
516
+ * @param {{ messageTokenBudget?: number, recentTurnCap?: number, maxMessageCount?: number, keepToolTurns?: number, language?: string }} [options]
517
+ * @returns {Array<object>}
518
+ */
519
+ export function trimSnapshotForBudget(snapshot, options = {}) {
520
+ if (!Array.isArray(snapshot) || snapshot.length === 0) return [];
521
+
522
+ const recentTurnCap = Number.isFinite(options.recentTurnCap) && options.recentTurnCap > 0
523
+ ? Math.floor(options.recentTurnCap)
524
+ : DEFAULT_RECENT_TURN_CAP;
525
+ const messageTokenBudget = Number.isFinite(options.messageTokenBudget) && options.messageTokenBudget > 0
526
+ ? Math.floor(options.messageTokenBudget)
527
+ : DEFAULT_MESSAGE_TOKEN_BUDGET;
528
+ const maxMessageCount = Number.isFinite(options.maxMessageCount) && options.maxMessageCount > 0
529
+ ? Math.floor(options.maxMessageCount)
530
+ : DEFAULT_RUNTIME_CACHE_MESSAGE_CAP;
531
+
532
+ let trimmed = sliceLastNTurns(snapshot, recentTurnCap);
533
+ if (trimmed.length > maxMessageCount) trimmed = trimmed.slice(-maxMessageCount);
534
+ let remainingTurnCap = recentTurnCap;
535
+ let tokens = estimateMessagesTokens(trimmed);
536
+ while (tokens > messageTokenBudget && remainingTurnCap > 1) {
537
+ const nextTurnCap = remainingTurnCap - 1;
538
+ const next = sliceLastNTurns(trimmed, nextTurnCap);
539
+ if (next.length === trimmed.length) break;
540
+ remainingTurnCap = nextTurnCap;
541
+ trimmed = next;
542
+ tokens = estimateMessagesTokens(trimmed);
543
+ }
544
+
545
+ trimmed = stripToolNoiseFromOlderTurns(trimmed, {
546
+ keepToolTurns: options.keepToolTurns,
547
+ });
548
+ trimmed = truncateToolResultsForModel(trimmed, { language: options.language });
549
+ trimmed = pairSanitize(trimmed);
550
+ return fitMessagesToBudget(trimmed, messageTokenBudget);
551
+ }
552
+
553
+ /**
554
+ * Bound the Session-level runtime history cache. This is deliberately stricter
555
+ * than the provider configuration: the cache is only a disposable source
556
+ * snapshot, while ConversationStore retains the complete transcript.
557
+ *
558
+ * @param {Array<object>} snapshot
559
+ * @param {{ language?: string }} [options]
560
+ * @returns {Array<object>}
561
+ */
562
+ export function trimHistoryCacheForRuntime(snapshot, options = {}) {
563
+ return trimSnapshotForBudget(snapshot, {
564
+ recentTurnCap: DEFAULT_RUNTIME_CACHE_TURN_CAP,
565
+ messageTokenBudget: DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET,
566
+ maxMessageCount: DEFAULT_RUNTIME_CACHE_MESSAGE_CAP,
567
+ keepToolTurns: DEFAULT_KEEP_TOOL_TURNS,
568
+ language: options.language,
569
+ });
570
+ }
@@ -163,7 +163,7 @@ export function classifyPolicyError(statusCode, responseBody = '', details = {})
163
163
  return new LLMPolicyError(signals.message, status, details);
164
164
  }
165
165
 
166
- /** Context too long error (413 or API-specific) — need compaction. */
166
+ /** Context too long error (413 or API-specific). */
167
167
  export class LLMContextError extends Error {
168
168
  constructor(message) {
169
169
  super(message);
@@ -534,7 +534,7 @@ export class AnthropicAdapter extends LLMAdapter {
534
534
  * Non-streaming call for side queries.
535
535
  *
536
536
  * task-327c: accepts `effort` for internal scenario-tagged calls
537
- * (consolidate/dream/recall/light). Guards mirror stream() — unsupported
537
+ * (dream/recall/light). Guards mirror stream() — unsupported
538
538
  * models silently drop the param. max_tokens auto-widens to budget+1024
539
539
  * when needed.
540
540
  */
@@ -184,7 +184,8 @@ export async function listProviderModels(providerId, { yeaftDir = null } = {}) {
184
184
  * that provider's numbers verbatim — the caller knew which gateway it
185
185
  * was talking to.
186
186
  * • Otherwise we take the MIN of every provider's `context` and `output`.
187
- * Context is a ceiling: under-shooting risks an early compact (bad);
187
+ * Context is a ceiling: under-shooting risks an unnecessarily small
188
+ * request window; over-shooting risks an LLMContextError mid-query (worse);
188
189
  * over-shooting risks an LLMContextError mid-query (worse). Min picks
189
190
  * the safer side. Users who know better can pin numbers explicitly via
190
191
  * `providers[].models[].contextWindow` in `~/.yeaft/config.json`.
@@ -521,7 +521,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
521
521
  // ─── Non-streaming call() ───────────────────────────────
522
522
 
523
523
  /**
524
- * Side-query (consolidate / dream / recall / light) entry point. Does
524
+ * Side-query (dream / recall / light) entry point. Does
525
525
  * NOT accept `onRawExchange` — these calls intentionally don't surface
526
526
  * in the user-facing debug panel. If a future product change wants to
527
527
  * expose them, mirror the stream() instrumentation. Parity with
@@ -147,9 +147,8 @@ export function filterEffortForModel(params, context = {}) {
147
147
  /**
148
148
  * task-715: last-line-of-defense pair sanitize at the wire.
149
149
  *
150
- * `pairSanitize` already runs in two upstream paths
151
- * (`conversation/persist.js#loadRecentBySession` and
152
- * `history-compact.js#compactHistory`), but the engine's main loop
150
+ * `pairSanitize` already runs in the persisted-history path
151
+ * (`conversation/persist.js#loadRecentBySession`), but the engine's main loop
153
152
  * mutates `conversationMessages` AFTER those — appending tool results
154
153
  * mid-loop, archiving bulky tool results into stubs, and (in failure
155
154
  * paths) potentially leaving an assistant `tool_use` whose matching
@@ -48,8 +48,8 @@ function addUsage(total, usage) {
48
48
  * reports once after its event stream finishes or aborts, and every
49
49
  * non-streaming side call reports once whether it succeeds or fails.
50
50
  *
51
- * Parent VP engines, sub-agent engines, Dream, compact, reflection, AMS, and
52
- * classifiers all reuse this adapter, so none need their own accounting hook.
51
+ * Parent VP engines, sub-agent engines, Dream, reflection, AMS, and classifiers
52
+ * all reuse this adapter, so none need their own accounting hook.
53
53
  */
54
54
  export class UsageAccountingAdapter extends LLMAdapter {
55
55
  #adapter;
@@ -3,9 +3,9 @@
3
3
  * slice so it can be safely fed to the LLM adapter.
4
4
  *
5
5
  * Why this exists:
6
- * `agent/yeaft/conversation/persist.js#loadRecentBySession` and
7
- * `agent/yeaft/history-compact.js#compactHistory` both produce
8
- * sub-slices of a longer message stream. Both paths can — depending on
6
+ * `agent/yeaft/conversation/persist.js#loadRecentBySession` and the
7
+ * deterministic provider history window both produce sub-slices of a longer
8
+ * message stream. Both paths can — depending on
9
9
  * where the cut lands — produce one of two illegal shapes:
10
10
  * 1. A `role: 'tool'` message whose owning assistant `tool_use` is
11
11
  * no longer in the slice.
package/yeaft/prompts.js CHANGED
@@ -16,10 +16,9 @@
16
16
  * ④ Active Scope — structured per-turn scope summary
17
17
  * (session / vp / members / envelope IDs)
18
18
  *
19
- * The compact summary, user_profile, and core_memory blocks that used to
20
- * live inside the system prompt are GONE. Compact summary is now part of
21
- * the messages timeline; user_profile + core_memory have been folded into
22
- * AMS Resident.
19
+ * Long-term semantic context comes only from the AMS Memory outlet. The
20
+ * conversation transcript stays in the messages timeline and is bounded by
21
+ * deterministic per-request history-window trimming.
23
22
  */
24
23
 
25
24
  import { readFileSync, existsSync } from 'fs';