@yeaft/webchat-agent 1.0.446 → 1.0.448

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/local-runtime/server/context.js +3 -22
  2. package/local-runtime/server/database.js +1 -0
  3. package/local-runtime/server/db/agent-inventory-db.js +122 -0
  4. package/local-runtime/server/db/connection.js +59 -0
  5. package/local-runtime/server/handlers/agent-sync.js +12 -1
  6. package/local-runtime/server/handlers/client-conversation.js +1 -0
  7. package/local-runtime/server/routes/admin-routes.js +78 -15
  8. package/local-runtime/server/ws-agent.js +41 -3
  9. package/local-runtime/server/ws-client.js +1 -7
  10. package/local-runtime/version.json +1 -1
  11. package/local-runtime/web/app.bundle.js +208 -233
  12. package/local-runtime/web/app.bundle.js.gz +0 -0
  13. package/local-runtime/web/index.html +2 -2
  14. package/local-runtime/web/style.bundle.css +1 -1
  15. package/local-runtime/web/style.bundle.css.gz +0 -0
  16. package/package.json +1 -1
  17. package/yeaft/archive/turn-archive.js +1 -1
  18. package/yeaft/cli.js +15 -17
  19. package/yeaft/config.js +2 -3
  20. package/yeaft/conversation/internal-control.js +2 -0
  21. package/yeaft/conversation/persist.js +22 -216
  22. package/yeaft/conversation/visible-entry.js +1 -1
  23. package/yeaft/effort.js +0 -2
  24. package/yeaft/engine.js +31 -376
  25. package/yeaft/history-window.js +570 -0
  26. package/yeaft/llm/adapter.js +1 -1
  27. package/yeaft/llm/anthropic.js +1 -1
  28. package/yeaft/llm/models-dev.js +2 -1
  29. package/yeaft/llm/openai-responses.js +1 -1
  30. package/yeaft/llm/router.js +2 -3
  31. package/yeaft/llm/usage-accounting.js +2 -2
  32. package/yeaft/pair-sanitize.js +3 -3
  33. package/yeaft/prompts.js +3 -4
  34. package/yeaft/session.js +0 -51
  35. package/yeaft/stdio-protocol.js +0 -1
  36. package/yeaft/stop-hooks.js +5 -6
  37. package/yeaft/turn-utils.js +5 -5
  38. package/yeaft/web-bridge.js +57 -135
  39. package/yeaft/compact/compactor.js +0 -283
  40. package/yeaft/compact/orchestrator.js +0 -141
  41. package/yeaft/compact/partition.js +0 -94
  42. package/yeaft/compact/triggers.js +0 -54
  43. package/yeaft/compact/turn-group.js +0 -85
  44. package/yeaft/history-compact.js +0 -764
@@ -1,764 +0,0 @@
1
- /**
2
- * history-compact.js — In-memory conversation history compaction for the
3
- * Yeaft group-chat fan-out path.
4
- *
5
- * Problem this solves:
6
- * `agent/yeaft/web-bridge.js` keeps a flat module-level array
7
- * `conversationMessages` that grows unbounded across the lifetime of the
8
- * agent process. Every fan-out turn snapshots the whole thing into
9
- * `baseSnapshot` and feeds it to `engine.query` for every VP. Without a
10
- * cap, prompt size and token cost grow linearly with conversation length.
11
- *
12
- * Existing infrastructure (`agent/yeaft/compact/orchestrator.js`,
13
- * `engine.js#runOrchestratorCompact`) compacts the on-disk
14
- * `conversationStore` — a different surface. This helper compacts the
15
- * in-memory array that actually gets passed to the LLM.
16
- *
17
- * Approach (Claude-Code-style compact):
18
- * 1. Skip tool messages and the synthetic `_reflection`/`_compactSummary`
19
- * wrappers when feeding the summarizer (tool result bodies are noise;
20
- * reflection wrappers are already a summary).
21
- * 2. Ask the fast model to produce a short structured summary of the
22
- * conversation up to a cut-point.
23
- * 3. Replace `messages[0..cutIdx]` with ONE synthetic user message
24
- * carrying that summary, wrapped with the canonical recovery prompt
25
- * ("This session is being continued from a previous conversation...").
26
- * 4. Keep the last `keepRecent` user→assistant turns intact so the model
27
- * has fresh, untransformed context for whatever the user just said.
28
- *
29
- * Triggers:
30
- * - tokens < 12_000 → never compact (cheap chat, no point paying
31
- * the summarizer)
32
- * - fewer than 5 turns → do not compact unless context pressure is
33
- * already high
34
- * - tokens > 80 % of `maxContextTokens` (defaults to 200K → 160K)
35
- * - tokens > 200,000 hard ceiling
36
- *
37
- * The "turn > 20" trigger that an earlier revision used was dropped:
38
- * under a 30K token floor it's effectively dead code — the fractional
39
- * threshold fires first in any conversation big enough to matter.
40
- *
41
- * Defaults are derived from `maxContextTokens` so the policy auto-adjusts
42
- * when the user widens or narrows their context budget. All knobs are
43
- * overridable via the options bag for tests / future config plumbing.
44
- *
45
- * Why role='user' for the summary message:
46
- * The Anthropic Messages API rejects assistant prefill at the tail
47
- * ("messages must end with user before next assistant turn"). Wrapping
48
- * as user mirrors what Claude Code does for compact summaries — and
49
- * what `tool-folding/index.js#collapseRangeToReflection` already does
50
- * for tool-arc reflections in this codebase. The opening sentence
51
- * ("This session is being continued ...") makes the model treat it
52
- * as a recovery directive rather than a fresh user prompt.
53
- */
54
-
55
- import { estimateTokens } from './conversation/persist.js';
56
- import { pairSanitize } from './pair-sanitize.js';
57
- import { truncateToolResultIfNeeded } from './tools/registry.js';
58
- import {
59
- countTurns as countTurnsImpl,
60
- indexOfNthTurnFromEnd,
61
- sliceLastNTurns,
62
- } from './turn-utils.js';
63
-
64
-
65
- function truncateToolResultsForModel(messages, opts = {}) {
66
- if (!Array.isArray(messages) || messages.length === 0) return [];
67
- return messages.map((m) => {
68
- if (!m || m.role !== 'tool' || typeof m.content !== 'string') return { ...m };
69
- return {
70
- ...m,
71
- content: truncateToolResultIfNeeded(m.content, {
72
- toolName: m.name || m.toolName || 'tool_result',
73
- language: opts.language,
74
- }),
75
- };
76
- });
77
- }
78
-
79
- /**
80
- * Re-export `countTurns` so existing callers / tests that import it
81
- * from this module continue to work. Implementation now lives in
82
- * `turn-utils.js` and is shared with `ConversationStore`.
83
- */
84
- export const countTurns = countTurnsImpl;
85
-
86
- /**
87
- * Default trigger thresholds (2026-05-02 policy update):
88
- * - never compact while total tokens < 12K (soft floor — most short
89
- * conversations under that aren't worth paying the summarizer
90
- * cost; the LLM hasn't started feeling the context yet either),
91
- * - otherwise compact if ANY of:
92
- * turnCount > 30 (back-stop for chats with many small turns)
93
- * tokens > 80 % of `maxContextTokens` (default 200K → 160K)
94
- * tokens > 200K hard ceiling
95
- * Fewer than 5 turns are protected from compact unless the token
96
- * threshold is already crossed.
97
- *
98
- * Lowered from 30K → 12K and re-enabled a turn-count back-stop because
99
- * the previous "soft floor of 30K, no turn cap" combination is dead in
100
- * the multi-VP fan-out path: hundreds of small turns happily stay below
101
- * 30K and never trigger compact, then `runVpTurn` feeds the whole 720+
102
- * message snapshot to the LLM and trips the provider's context window.
103
- * The snapshot trim in `web-bridge.js#trimSnapshotForBudget` is the
104
- * primary defense; this is the second-line trigger that compresses
105
- * the on-array form so subsequent turns also stay bounded.
106
- *
107
- * `turnLimit` and the `turn_count` reason code are still overridable
108
- * for tests / future config.
109
- *
110
- * Token thresholds are derived from `maxContextTokens` at evaluation
111
- * time so the policy auto-adjusts to the user's configured context.
112
- */
113
- export const DEFAULT_TURN_LIMIT = Infinity;
114
- export const DEFAULT_MIN_TOKEN_FLOOR = 0;
115
- export const DEFAULT_MAX_CONTEXT_TOKENS = 200_000;
116
- export const DEFAULT_TOKEN_FRACTION = 0.5;
117
- export const DEFAULT_HARD_TOKEN_CEILING = Infinity;
118
- export const DEFAULT_MIN_TURNS_FOR_COMPACT = 0;
119
- export const DEFAULT_KEEP_TOOL_TURNS = 3;
120
- export const DEFAULT_TOOL_CALL_COMPACT_THRESHOLD = 30;
121
- /**
122
- * Effective default token trigger when no `maxContextTokens` is provided:
123
- * min(80% of 200K, 200K) = 160K. Preserved as `DEFAULT_TOKEN_LIMIT` for
124
- * back-compat with existing tests that import this name.
125
- */
126
- export const DEFAULT_TOKEN_LIMIT = Math.min(
127
- Math.floor(DEFAULT_MAX_CONTEXT_TOKENS * DEFAULT_TOKEN_FRACTION),
128
- DEFAULT_HARD_TOKEN_CEILING
129
- );
130
-
131
- /**
132
- * How many user→assistant pairs to leave intact at the tail. The summary
133
- * replaces everything before this window. 2 keeps "what we were just
134
- * talking about" lossless.
135
- */
136
- export const DEFAULT_KEEP_RECENT_TURNS = 3;
137
-
138
- /**
139
- * Default cap on the number of turns kept in the per-call snapshot fed
140
- * to `engine.query` (see `trimSnapshotForBudget` below). A turn here is
141
- * one user-side prompt — multi-VP `@vp-X` variants of the same prompt
142
- * collapse into one turn (see `turn-utils.js#countTurns`).
143
- *
144
- * Sized in conjunction with `DEFAULT_TURN_LIMIT` (30, the compact-trigger
145
- * back-stop): trim to 25 leaves a 5-turn buffer below the compact trigger
146
- * so a typical chat sees its history compacted before the trim starts
147
- * dropping turns silently. That ordering matters — compact preserves the
148
- * tail's lossless 2 turns AND a summary of everything older, whereas trim
149
- * just discards anything beyond the cap.
150
- *
151
- * 25 turns at ~5 messages each (user + assistant + a couple tool steps)
152
- * is roughly 100–125 messages — well under the LLM context window for
153
- * any reasonable model, and large enough to preserve "what we've been
154
- * talking about" context for the model. The hard token-budget cap inside
155
- * `trimSnapshotForBudget` tightens this further when individual turns
156
- * are large.
157
- */
158
- export const DEFAULT_RECENT_TURN_CAP = 25;
159
-
160
- /**
161
- * Default per-query token budget for the snapshot (separate from the
162
- * `tokenLimit` used by compact triggers). Mirrors the default
163
- * carried in `~/.yeaft/config.json`'s `messageTokenBudget` field.
164
- */
165
- export const DEFAULT_MESSAGE_TOKEN_BUDGET = 32768;
166
-
167
- /**
168
- * Estimate the token weight of a single message including role overhead
169
- * and any tool-call structure. Mirrors `dream/segment.js` approach: a
170
- * couple of tokens per message for role/wrapping plus the body.
171
- *
172
- * @param {{role:string, content?:string, toolCalls?:Array, toolCallId?:string}} m
173
- * @returns {number}
174
- */
175
- export function estimateMessageTokens(m) {
176
- if (!m || typeof m !== 'object') return 0;
177
- let n = 2; // role + framing
178
- if (typeof m.content === 'string') n += estimateTokens(m.content);
179
- if (Array.isArray(m.toolCalls)) {
180
- for (const tc of m.toolCalls) {
181
- n += 4; // call framing
182
- try {
183
- const inputJson = typeof tc.input === 'string' ? tc.input : JSON.stringify(tc.input || {});
184
- n += estimateTokens(inputJson);
185
- } catch { /* ignore — JSON.stringify failure on circular input */ }
186
- if (tc.name) n += estimateTokens(tc.name);
187
- }
188
- }
189
- if (m.toolCallId) n += 2;
190
- return n;
191
- }
192
-
193
- /**
194
- * Sum estimated tokens across all messages.
195
- * @param {Array<object>} messages
196
- * @returns {number}
197
- */
198
- export function estimateMessagesTokens(messages) {
199
- if (!Array.isArray(messages)) return 0;
200
- let total = 0;
201
- for (const m of messages) total += estimateMessageTokens(m);
202
- return total;
203
- }
204
-
205
- /**
206
- * Pure trigger evaluator. Decides whether the in-memory history needs
207
- * compaction. No I/O, no LLM call.
208
- *
209
- * Policy (2026-05-22):
210
- * 1. tokens < `minTokenFloor` (default 12K) → trigger=false (always).
211
- * 2. fewer than `minTurnsForCompact` turns (default 5) → trigger=false
212
- * unless tokenCount already exceeds the fractional context threshold.
213
- * 3. otherwise trigger if ANY of:
214
- * turnCount > turnLimit (default 30 back-stop)
215
- * tokenCount > maxContextTokens*fraction (default 80%, reason='token_threshold')
216
- * tokenCount > hardTokenCeiling (reason='token_ceiling')
217
- *
218
- * `tokenLimit` is preserved as a back-compat override for callers /
219
- * tests that pin a specific number; when set, it overrides the
220
- * fraction-of-context calculation.
221
- *
222
- * @param {Array<object>} messages
223
- * @param {{
224
- * turnLimit?: number,
225
- * minTurnsForCompact?: number,
226
- * tokenLimit?: number,
227
- * minTokenFloor?: number,
228
- * maxContextTokens?: number,
229
- * tokenFraction?: number,
230
- * hardTokenCeiling?: number,
231
- * }} [opts]
232
- * @returns {{trigger: boolean, reason: 'turn_count'|'token_threshold'|'token_ceiling'|null,
233
- * turnCount: number, tokenCount: number,
234
- * turnLimit: number, tokenLimit: number, minTurnsForCompact: number,
235
- * minTokenFloor: number, hardTokenCeiling: number}}
236
- */
237
- export function shouldCompactHistory(messages, opts = {}) {
238
- const turnLimit = opts.turnLimit ?? DEFAULT_TURN_LIMIT;
239
- const minTokenFloor = opts.minTokenFloor ?? DEFAULT_MIN_TOKEN_FLOOR;
240
- const hardTokenCeiling = opts.hardTokenCeiling ?? DEFAULT_HARD_TOKEN_CEILING;
241
- const maxContextTokens = opts.maxContextTokens ?? DEFAULT_MAX_CONTEXT_TOKENS;
242
- const tokenFraction = opts.tokenFraction ?? DEFAULT_TOKEN_FRACTION;
243
- const minTurnsForCompact = opts.minTurnsForCompact ?? DEFAULT_MIN_TURNS_FOR_COMPACT;
244
- // tokenLimit override wins; otherwise compute fractional threshold.
245
- const tokenLimit =
246
- opts.tokenLimit
247
- ?? Math.min(Math.floor(maxContextTokens * tokenFraction), hardTokenCeiling);
248
-
249
- const turnCount = countTurns(messages);
250
- const tokenCount = estimateMessagesTokens(messages);
251
-
252
- let reason = null;
253
- // Product rule: async group compact is allowed only when the current
254
- // conversation exceeds the model context window threshold. Turn count is
255
- // preserved as an explicit test/future-config override, but defaults to
256
- // Infinity so it cannot compact a small context by itself.
257
- if (tokenCount < minTokenFloor || (turnCount < minTurnsForCompact && tokenCount < tokenLimit)) {
258
- return {
259
- trigger: false,
260
- reason: null,
261
- turnCount,
262
- tokenCount,
263
- turnLimit,
264
- tokenLimit,
265
- minTurnsForCompact,
266
- minTokenFloor,
267
- hardTokenCeiling,
268
- };
269
- }
270
- if (tokenCount > hardTokenCeiling) reason = 'token_ceiling';
271
- else if (tokenCount > tokenLimit) reason = 'token_threshold';
272
-
273
- return {
274
- trigger: reason !== null,
275
- reason,
276
- turnCount,
277
- tokenCount,
278
- turnLimit,
279
- tokenLimit,
280
- minTurnsForCompact,
281
- minTokenFloor,
282
- hardTokenCeiling,
283
- };
284
- }
285
-
286
- function hasContentAfterToolStrip(content) {
287
- if (typeof content === 'string') return content.trim().length > 0;
288
- if (Array.isArray(content)) return content.length > 0;
289
- return content != null;
290
- }
291
-
292
- function countToolCallsInContent(content) {
293
- if (!Array.isArray(content)) return 0;
294
- let n = 0;
295
- for (const part of content) {
296
- if (!part || typeof part !== 'object') continue;
297
- if (part.type === 'tool_use' || part.type === 'function_call') n++;
298
- }
299
- return n;
300
- }
301
-
302
- function countToolCallsInMessages(messages) {
303
- if (!Array.isArray(messages)) return 0;
304
- let n = 0;
305
- for (const m of messages) {
306
- if (!m || typeof m !== 'object') continue;
307
- if (Array.isArray(m.toolCalls)) n += m.toolCalls.length;
308
- n += countToolCallsInContent(m.content);
309
- }
310
- return n;
311
- }
312
-
313
- function stripToolContentParts(content) {
314
- if (!Array.isArray(content)) return content;
315
- return content.filter(part => {
316
- if (!part || typeof part !== 'object') return true;
317
- return part.type !== 'tool_use'
318
- && part.type !== 'tool_result'
319
- && part.type !== 'function_call'
320
- && part.type !== 'function_call_output';
321
- });
322
- }
323
-
324
- /**
325
- * Remove tool-call / tool-result noise from turns older than the recent
326
- * lossless window. The last `keepToolTurns` turns keep their full tool
327
- * chains; older turns keep user/assistant text but lose `toolCalls`,
328
- * Anthropic/OpenAI tool content blocks, and `role:'tool'` messages.
329
- *
330
- * This is deliberately a wire-history transform, not a summarizer: it
331
- * never invents a summary and it never mutates input. Pair-sanitize runs
332
- * afterwards so no orphan tool_use/tool_result can survive.
333
- *
334
- * @param {Array<object>} messages
335
- * @param {{ keepToolTurns?: number }} [opts]
336
- * @returns {Array<object>}
337
- */
338
- export function stripToolNoiseFromOlderTurns(messages, opts = {}) {
339
- if (!Array.isArray(messages) || messages.length === 0) return [];
340
- const keepToolTurns = Number.isFinite(opts.keepToolTurns) && opts.keepToolTurns >= 0
341
- ? opts.keepToolTurns
342
- : DEFAULT_KEEP_TOOL_TURNS;
343
- const cutIdx = indexOfNthTurnFromEnd(messages, keepToolTurns);
344
- if (cutIdx <= 0) return messages.map(m => ({ ...m }));
345
-
346
- const older = messages.slice(0, cutIdx);
347
- const recent = messages.slice(cutIdx);
348
- const cleanedOlder = [];
349
-
350
- for (const m of older) {
351
- if (!m || typeof m !== 'object') continue;
352
- if (m.role === 'tool') continue;
353
-
354
- const next = { ...m };
355
- if (Array.isArray(next.toolCalls)) delete next.toolCalls;
356
- if (Array.isArray(next.content)) next.content = stripToolContentParts(next.content);
357
-
358
- if (next.role === 'assistant' && !hasContentAfterToolStrip(next.content)) continue;
359
- if (next.role === 'user' && Array.isArray(next.content) && next.content.length === 0) continue;
360
- cleanedOlder.push(next);
361
- }
362
-
363
- return [...cleanedOlder, ...recent.map(m => ({ ...m }))];
364
- }
365
-
366
- /**
367
- * Apply the async compact retained-tail tool policy. Small retained tails keep
368
- * every tool pair intact. Once the retained tail exceeds the threshold, keep
369
- * full tool history only for the latest turn and strip tool noise from the
370
- * earlier retained turns while preserving their normal text.
371
- *
372
- * @param {Array<object>} tail
373
- * @param {{ keepToolTurns?: number, toolCallCompactThreshold?: number }} [opts]
374
- * @returns {Array<object>}
375
- */
376
- export function compactRetainedTailToolCalls(tail, opts = {}) {
377
- if (!Array.isArray(tail) || tail.length === 0) return [];
378
-
379
- const threshold = Number.isFinite(opts.toolCallCompactThreshold) && opts.toolCallCompactThreshold >= 0
380
- ? opts.toolCallCompactThreshold
381
- : DEFAULT_TOOL_CALL_COMPACT_THRESHOLD;
382
- const toolCallCount = countToolCallsInMessages(tail);
383
- if (toolCallCount <= threshold) return tail.map(m => ({ ...m }));
384
-
385
- const keepToolTurns = Number.isFinite(opts.keepToolTurns) && opts.keepToolTurns >= 0
386
- ? opts.keepToolTurns
387
- : 1;
388
- return stripToolNoiseFromOlderTurns(tail, { keepToolTurns });
389
- }
390
-
391
- /**
392
- * Strip noise from a message list before sending it to the summarizer:
393
- * - drop `role: 'tool'` (raw tool results — too verbose, mostly redundant)
394
- * - drop messages already tagged `_compactSummary` (avoid summarising
395
- * a summary)
396
- * - keep `_reflection` messages as-is (they're already a fold-summary
397
- * of an earlier tool arc and contain real information)
398
- * - elide `toolCalls` from assistant messages: replace each with a tag
399
- * line like "[called tool: bash with input ...]" so the summarizer
400
- * knows a tool ran without spending tokens on the full input
401
- *
402
- * @param {Array<object>} messages
403
- * @returns {Array<{role:string, content:string}>}
404
- */
405
- export function buildSummarizerInput(messages) {
406
- if (!Array.isArray(messages)) return [];
407
- const out = [];
408
- for (const m of messages) {
409
- if (!m || typeof m !== 'object') continue;
410
- if (m.role === 'tool') continue;
411
- if (m._compactSummary) continue;
412
- let content = typeof m.content === 'string' ? m.content : '';
413
- if (m.role === 'assistant' && Array.isArray(m.toolCalls) && m.toolCalls.length > 0) {
414
- const callTags = m.toolCalls.map(tc => {
415
- const name = tc.name || 'unknown';
416
- let inputBrief = '';
417
- try {
418
- const json = typeof tc.input === 'string' ? tc.input : JSON.stringify(tc.input || {});
419
- inputBrief = json.length > 120 ? json.slice(0, 120) + '…' : json;
420
- } catch { inputBrief = '<input>'; }
421
- return `[tool ${name}: ${inputBrief}]`;
422
- }).join(' ');
423
- content = content ? `${content}\n${callTags}` : callTags;
424
- }
425
- if (!content) continue;
426
- out.push({ role: m.role, content });
427
- }
428
- return out;
429
- }
430
-
431
- /**
432
- * Find the cut index: keep the last `keepRecent` user→assistant arcs
433
- * intact, fold everything before. Returns the index that the cut starts
434
- * AT, i.e. messages[0..cutIdx) gets summarised, messages[cutIdx..] stays.
435
- *
436
- * Thin wrapper around `turn-utils.indexOfNthTurnFromEnd` with the
437
- * historical contract preserved:
438
- * - empty input returns -1,
439
- * - `keepRecent <= 0` folds everything (returns messages.length),
440
- * - "fewer turns than keepRecent" maps to -1 (caller treats as no-op).
441
- *
442
- * Multi-VP fan-out: `@vp-X` variants of the same underlying turn count
443
- * as ONE turn and the boundary extends backwards through them all.
444
- *
445
- * @param {Array<object>} messages
446
- * @param {number} keepRecent
447
- * @returns {number}
448
- */
449
- export function findCutIndex(messages, keepRecent) {
450
- if (!Array.isArray(messages) || messages.length === 0) return -1;
451
- if (keepRecent <= 0) return messages.length; // fold everything
452
- const idx = indexOfNthTurnFromEnd(messages, keepRecent);
453
- // `indexOfNthTurnFromEnd` returns -1 when there are fewer turns than
454
- // requested — historical contract is the same. Pass through.
455
- return idx;
456
- }
457
-
458
- /**
459
- * Wrap a summary string into the canonical "session continued" recovery
460
- * message. The wording is deliberately close to Claude Code's compact
461
- * marker so frontend filters (already in `web/stores/helpers/assistantOutput.js`,
462
- * `server/db/message-db.js`) recognise it.
463
- *
464
- * @param {string} summary
465
- * @param {{ language?: string }} [opts]
466
- * @returns {{role:'user', content:string, _compactSummary: true}}
467
- */
468
- export function wrapSummaryAsUserMessage(summary, opts = {}) {
469
- const body = (summary || '').trim() || '(no summary produced)';
470
- const isZh = String(opts.language || '').toLowerCase().startsWith('zh');
471
- const content = isZh
472
- ? '本会话延续自之前的对话。早期上下文已经被概括以节省空间。\n\n' +
473
- '至此为止的对话摘要:\n' +
474
- body +
475
- '\n\n请从中断处继续对话,不要再向用户重复确认。'
476
- : 'This session is being continued from a previous conversation. ' +
477
- 'The earlier context has been summarized for efficiency.\n\n' +
478
- 'Summary of conversation so far:\n' +
479
- body +
480
- '\n\nContinue the conversation from where it left off without asking the user any further questions.';
481
- return {
482
- role: 'user',
483
- content,
484
- _compactSummary: true,
485
- };
486
- }
487
-
488
- /**
489
- * Build the prompt fed to the fast-model summarizer. Kept in code (not in
490
- * a template file) because it's small and lives alongside the call site.
491
- *
492
- * The summarizer prompt itself is language-aware: callers pass the live
493
- * `config.language` so the produced summary is written in the user's
494
- * preferred language. JSON-style structural cues stay English so the
495
- * summary remains easy to splice into the next turn regardless of locale.
496
- *
497
- * @param {Array<{role:string, content:string}>} cleanedMessages
498
- * @param {{ language?: string }} [opts]
499
- * @returns {{system: string, prompt: string}}
500
- */
501
- export function buildSummaryPrompt(cleanedMessages, opts = {}) {
502
- const transcript = cleanedMessages
503
- .map(m => `[${m.role}]\n${m.content}`)
504
- .join('\n\n---\n\n');
505
- const isZh = String(opts.language || '').toLowerCase().startsWith('zh');
506
- const system = isZh
507
- ? '你是多 agent 群聊的对话摘要器。请用中文写出 4–8 条简明 bullet 摘要。' +
508
- '保留:(1) 已做的决策,(2) 已学到的事实,(3) 用户当前目标,' +
509
- '(4) 任何未解决的问题或待办事项,(5) 哪些 VP 参与了对话以及各自贡献。' +
510
- '不要包含原始工具输出。不要臆测。要具体。'
511
- : 'You are a conversation summarizer for a multi-agent group chat. ' +
512
- 'Produce a concise (4–8 short bullet points) summary of the conversation ' +
513
- 'so far. Preserve: (1) decisions made, (2) facts learned, (3) the user\'s ' +
514
- 'current goal, (4) any open questions or pending actions, (5) which VPs ' +
515
- 'are participating and what each contributed. Do NOT include raw tool ' +
516
- 'output. Do NOT speculate. Be specific.';
517
- const prompt = isZh
518
- ? '请概括下面的对话。只输出摘要正文,不要前言。\n\n' + transcript
519
- : 'Summarize the following conversation. Output ONLY the summary, no ' +
520
- 'preamble.\n\n' +
521
- transcript;
522
- return { system, prompt };
523
- }
524
-
525
- /**
526
- * Apply compaction to a messages array. Pure transform once `summarize`
527
- * has produced text. Returns a new array — does not mutate the input.
528
- *
529
- * @param {Array<object>} messages
530
- * @param {{
531
- * summarize: (args: {system: string, prompt: string}) => Promise<string>,
532
- * keepRecent?: number,
533
- * turnLimit?: number,
534
- * tokenLimit?: number,
535
- * minTokenFloor?: number,
536
- * maxContextTokens?: number,
537
- * tokenFraction?: number,
538
- * hardTokenCeiling?: number,
539
- * language?: string,
540
- * }} options
541
- * @returns {Promise<{
542
- * messages: Array<object>,
543
- * compacted: boolean,
544
- * reason: string|null,
545
- * summary: string|null,
546
- * archivedCount: number,
547
- * beforeTurns: number,
548
- * beforeTokens: number,
549
- * afterTurns: number,
550
- * afterTokens: number,
551
- * }>}
552
- */
553
- export async function compactHistory(messages, options) {
554
- const {
555
- summarize,
556
- keepRecent = DEFAULT_KEEP_RECENT_TURNS,
557
- turnLimit,
558
- tokenLimit,
559
- minTokenFloor,
560
- maxContextTokens,
561
- tokenFraction,
562
- hardTokenCeiling,
563
- language,
564
- keepToolTurns,
565
- toolCallCompactThreshold,
566
- } = options || {};
567
-
568
- if (typeof summarize !== 'function') {
569
- throw new TypeError('compactHistory: options.summarize must be a function');
570
- }
571
-
572
- // Pass thresholds through to shouldCompactHistory so a single options
573
- // bag controls the policy. Undefined keys fall back to module defaults.
574
- const triggerOpts = {
575
- turnLimit,
576
- tokenLimit,
577
- minTokenFloor,
578
- maxContextTokens,
579
- tokenFraction,
580
- hardTokenCeiling,
581
- minTurnsForCompact: options?.minTurnsForCompact,
582
- };
583
- const before = shouldCompactHistory(messages, triggerOpts);
584
- if (!before.trigger) {
585
- return {
586
- messages,
587
- compacted: false,
588
- reason: null,
589
- summary: null,
590
- archivedCount: 0,
591
- beforeTurns: before.turnCount,
592
- beforeTokens: before.tokenCount,
593
- afterTurns: before.turnCount,
594
- afterTokens: before.tokenCount,
595
- };
596
- }
597
-
598
- const cutIdx = findCutIndex(messages, keepRecent);
599
- if (cutIdx <= 0) {
600
- // Not enough history to fold while preserving the recent window.
601
- return {
602
- messages,
603
- compacted: false,
604
- reason: before.reason,
605
- summary: null,
606
- archivedCount: 0,
607
- beforeTurns: before.turnCount,
608
- beforeTokens: before.tokenCount,
609
- afterTurns: before.turnCount,
610
- afterTokens: before.tokenCount,
611
- };
612
- }
613
-
614
- const archived = messages.slice(0, cutIdx);
615
- const tail = messages.slice(cutIdx);
616
- const cleaned = buildSummarizerInput(archived);
617
-
618
- let summaryText = '';
619
- if (cleaned.length > 0) {
620
- const { system, prompt } = buildSummaryPrompt(cleaned, { language });
621
- try {
622
- summaryText = (await summarize({ system, prompt })) || '';
623
- } catch (err) {
624
- // Summarizer failure → return original messages, signal failure.
625
- return {
626
- messages,
627
- compacted: false,
628
- reason: before.reason,
629
- summary: null,
630
- archivedCount: 0,
631
- beforeTurns: before.turnCount,
632
- beforeTokens: before.tokenCount,
633
- afterTurns: before.turnCount,
634
- afterTokens: before.tokenCount,
635
- error: err && err.message ? err.message : String(err),
636
- };
637
- }
638
- // Treat an empty / whitespace-only summary as a soft failure rather
639
- // than a successful compact. Otherwise we'd archive real history
640
- // behind a "(no summary produced)" placeholder and the next turn
641
- // would start from useless context.
642
- if (!summaryText.trim()) {
643
- return {
644
- messages,
645
- compacted: false,
646
- reason: before.reason,
647
- summary: null,
648
- archivedCount: 0,
649
- beforeTurns: before.turnCount,
650
- beforeTokens: before.tokenCount,
651
- afterTurns: before.turnCount,
652
- afterTokens: before.tokenCount,
653
- error: 'empty summary',
654
- };
655
- }
656
- }
657
-
658
- const summaryMsg = wrapSummaryAsUserMessage(summaryText, { language });
659
-
660
- // Defensive pair-sanitize: the cut at `cutIdx` lands at a user-message
661
- // boundary so an `[assistant(toolCalls), tool…]` arc is not split, but
662
- // we still run `pairSanitize` over the tail as belt-and-suspenders —
663
- // it idempotently drops any orphan tool messages, and any assistant
664
- // whose tool_use IDs aren't fully matched in the tail. This is what
665
- // keeps the next adapter call from 400-ing on tool_use/tool_result
666
- // mismatch when the storage / fan-out layer reorders messages.
667
- const compactedTail = compactRetainedTailToolCalls(tail, {
668
- keepToolTurns,
669
- toolCallCompactThreshold,
670
- });
671
- const safeTail = pairSanitize(compactedTail);
672
-
673
- const newMessages = [summaryMsg, ...safeTail];
674
- const after = shouldCompactHistory(newMessages, triggerOpts);
675
-
676
- return {
677
- messages: newMessages,
678
- compacted: true,
679
- reason: before.reason,
680
- summary: summaryText,
681
- archivedCount: archived.length,
682
- beforeTurns: before.turnCount,
683
- beforeTokens: before.tokenCount,
684
- afterTurns: after.turnCount,
685
- afterTokens: after.tokenCount,
686
- };
687
- }
688
-
689
- /**
690
- * Trim a snapshot of conversation messages so the per-call array fed
691
- * to `engine.query` stays bounded.
692
- *
693
- * Two-stage policy:
694
- * 1. **Turn cap** — keep at most `recentTurnCap` turns (default 25)
695
- * via `sliceLastNTurns`. This always cuts at a user-message
696
- * boundary and walks forward through `@vp-X` variants of the
697
- * cut turn so the slice is pair-safe.
698
- * 2. **Token budget** — if the trimmed slice still exceeds
699
- * `messageTokenBudget` tokens (default 32768 from
700
- * `~/.yeaft/config.json`), iteratively drop the oldest turn until
701
- * we're under budget. We never drop below 1 turn — even a single
702
- * huge turn is preferable to no context.
703
- *
704
- * Then run `pairSanitize` as belt-and-suspenders to drop any orphan
705
- * tool_use/tool_result that survived the cuts. The transform is
706
- * idempotent and never mutates the input.
707
- *
708
- * Why this exists:
709
- * `runVpTurn` previously fed the entire `conversationMessages` array
710
- * into `engine.query` for every fan-out. With multi-VP turns the
711
- * array grows ~5–8 messages per user prompt, so after a few hundred
712
- * prompts the per-call payload exceeds 100 KB and routinely OOMs the
713
- * provider's context window. `compactHistory` only fires above its
714
- * token soft floor — small chats with many turns stay below that
715
- * floor but still bloat the messages array. This trim is the second-
716
- * line defense: it ALWAYS runs, before every query, regardless of
717
- * compact state.
718
- *
719
- * Lives in `history-compact.js` alongside `compactHistory` because
720
- * both functions are part of the same "bound the messages array fed
721
- * to the LLM" surface — keeping them together makes the relationship
722
- * between trim (per-call) and compact (global) explicit.
723
- *
724
- * @param {Array<object>} snapshot
725
- * @param {{ messageTokenBudget?: number, recentTurnCap?: number, keepToolTurns?: number, language?: string }} [opts]
726
- * @returns {Array<object>}
727
- */
728
- export function trimSnapshotForBudget(snapshot, opts = {}) {
729
- if (!Array.isArray(snapshot) || snapshot.length === 0) return [];
730
-
731
- const recentTurnCap = Number.isFinite(opts.recentTurnCap) && opts.recentTurnCap > 0
732
- ? opts.recentTurnCap
733
- : DEFAULT_RECENT_TURN_CAP;
734
- const messageTokenBudget = Number.isFinite(opts.messageTokenBudget) && opts.messageTokenBudget > 0
735
- ? opts.messageTokenBudget
736
- : DEFAULT_MESSAGE_TOKEN_BUDGET;
737
-
738
- // Stage 1: cap by turn count.
739
- let trimmed = sliceLastNTurns(snapshot, recentTurnCap);
740
-
741
- // Stage 2: cap by token budget. Drop oldest turn iteratively.
742
- // We never drop below ~1 turn — pick a safety floor of 1.
743
- let remainingTurnCap = recentTurnCap;
744
- let tokens = estimateMessagesTokens(trimmed);
745
- while (tokens > messageTokenBudget && remainingTurnCap > 1) {
746
- remainingTurnCap--;
747
- trimmed = sliceLastNTurns(trimmed, remainingTurnCap);
748
- tokens = estimateMessagesTokens(trimmed);
749
- }
750
-
751
- // Stage 3: keep only the recent tool chains lossless. Older turns
752
- // retain text but drop tool_use/tool_result noise before pair safety.
753
- trimmed = stripToolNoiseFromOlderTurns(trimmed, {
754
- keepToolTurns: opts.keepToolTurns,
755
- });
756
-
757
- // Stage 4: bound the raw tool result copy that is fed back into the model.
758
- // The in-memory/persisted transcript keeps the full content; this transform
759
- // only affects the per-query snapshot passed to engine.query().
760
- trimmed = truncateToolResultsForModel(trimmed, { language: opts.language });
761
-
762
- // Stage 5: pair-sanitize to drop orphan tool_use/tool_result.
763
- return pairSanitize(trimmed);
764
- }