@yeaft/webchat-agent 1.0.447 → 1.0.449

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/yeaft/engine.js CHANGED
@@ -3,13 +3,13 @@
3
3
  *
4
4
  * The engine is the core orchestrator:
5
5
  * 1. Before first turn: recall memories → inject into system prompt
6
- * 2. Build messages array (with compact summary if available)
6
+ * 2. Build messages array from persisted history and the current prompt
7
7
  * 3. Call adapter.stream()
8
8
  * 4. Collect text + tool_calls from stream events
9
9
  * 5. If tool_calls → execute tools → append results → goto 3
10
- * 6. Persist each completed message at its durability boundary; end_turn runs maintenance
10
+ * 6. Persist each completed message at its durability boundary
11
11
  * 7. If max_tokens → auto-continue (up to maxContinueTurns)
12
- * 8. On LLMContextError → force compact → retry
12
+ * 8. On LLMContextError → fail the turn; no summary or hidden maintenance call
13
13
  * 9. On retryable error with fallbackModel → switch model → retry
14
14
  *
15
15
  * Pattern derived from Claude Code's query loop (src/query.ts).
@@ -22,7 +22,7 @@ import { promises as fsp } from 'fs';
22
22
  import { join, resolve as resolvePath } from 'path';
23
23
  import { buildSystemPrompt, buildWorkerPrompt } from './prompts.js';
24
24
  import { getRuntimePlatformInfo } from './runtime-platform.js';
25
- import { LLMContextError, LLMAbortError, LLMAuthError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
25
+ import { LLMAbortError, LLMAuthError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
26
26
  import { runMemoryPreflow, buildRelevantScopes, memoryScopeLabel } from './sessions/pre-flow.js';
27
27
  import {
28
28
  readProjectDoc,
@@ -32,11 +32,8 @@ import {
32
32
  projectDocWriteScopesNeedingReload,
33
33
  DEFAULT_PROJECT_DOC_MAX_BYTES,
34
34
  } from './sessions/project-doc.js';
35
- import { partitionMessages } from './compact/partition.js';
36
- import { runCompact as runCompactOrchestrator } from './compact/orchestrator.js';
37
- import { evaluateCompactTriggers } from './compact/triggers.js';
38
- import { archiveTurn } from './archive/turn-archive.js';
39
35
  import { archiveToolResults } from './archive/tool-results.js';
36
+ import { trimSnapshotForBudget } from './history-window.js';
40
37
  import { isVpForeign, readContent as readScopeContent } from './memory/store.js';
41
38
  import { ActiveMemorySet } from './memory/ams.js';
42
39
  import { cleanMemoryPromptText } from './memory/prompt-cleanup.js';
@@ -48,7 +45,6 @@ const MAIN_THREAD_ID = 'main';
48
45
  import { pickEffort, parseEffortPrefix } from './effort.js';
49
46
  import { DEFAULT_CONTEXT_WINDOW, normalizeEffort, resolveContextWindow, resolveModel } from './models.js';
50
47
  import { lookupModelLimitSync } from './llm/models-dev.js';
51
- import { countTurns } from './turn-utils.js';
52
48
  import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
53
49
  import { resolveThinking } from './router/thinking.js';
54
50
  import { approxTokens, computeBudget } from './memory/budget.js';
@@ -81,7 +77,7 @@ import {
81
77
  * conversations (user report: Yeaft loop errored at the cap). The engine
82
78
  * now runs until the LLM itself returns stopReason='end_turn' or a
83
79
  * non-retryable error surfaces. Real runaway loops are still bounded by:
84
- * • provider rate limits / context window (LLMContextError → compact)
80
+ * • provider rate limits / context window (LLMContextError is surfaced)
85
81
  * • user-initiated abort (AbortController / cancel)
86
82
  * • MAX_CONTINUE_TURNS for the max_tokens auto-continue path
87
83
  */
@@ -96,9 +92,6 @@ const MAX_CONTINUE_TURNS = 3;
96
92
  */
97
93
  const MAX_CONCURRENT_READ_ONLY_TOOLS = 4;
98
94
 
99
- /** Bound the best-effort post-turn AMS LLM call independently of the user turn. */
100
- const MAINTENANCE_CALL_TIMEOUT_MS = 30_000;
101
-
102
95
  /** Maximum silence while a visible turn waits for a result-producing task. */
103
96
  const DEFAULT_ASYNC_TASK_WAIT_TIMEOUT_MS = 120_000;
104
97
 
@@ -409,7 +402,6 @@ export function estimateMessagesTokens(system, messages) {
409
402
  }
410
403
 
411
404
  export const GROUP_CONTEXT_PRESSURE_RATIO = 0.8;
412
- export const GROUP_MIN_TURNS_FOR_COMPACT = 5;
413
405
 
414
406
  export function shouldAllowGroupReflection({
415
407
  system = '',
@@ -421,12 +413,10 @@ export function shouldAllowGroupReflection({
421
413
  if (!sessionId) {
422
414
  return {
423
415
  allowed: true,
424
- compactAllowed: true,
425
416
  tokenEstimate: estimateMessagesTokens(system, messages),
426
417
  threshold: 0,
427
418
  contextWindow: null,
428
419
  ratio: GROUP_CONTEXT_PRESSURE_RATIO,
429
- turnCount: countTurns(messages),
430
420
  usedFallbackContextWindow: false,
431
421
  };
432
422
  }
@@ -440,19 +430,14 @@ export function shouldAllowGroupReflection({
440
430
  const threshold = Math.floor(contextWindow * GROUP_CONTEXT_PRESSURE_RATIO);
441
431
  const tokenEstimate = estimateMessagesTokens(system, messages);
442
432
  const overThreshold = tokenEstimate >= threshold;
443
- const turnCount = countTurns(messages);
444
433
  return {
445
434
  // Group send defaults to no reflection. Trust the model until context
446
435
  // pressure says we are near the model window.
447
436
  allowed: overThreshold,
448
- // Durable compact is also protected for tiny histories: fewer than five
449
- // turns do not compact unless they already exceed the same 80% threshold.
450
- compactAllowed: overThreshold || turnCount >= GROUP_MIN_TURNS_FOR_COMPACT,
451
437
  tokenEstimate,
452
438
  threshold,
453
439
  contextWindow,
454
440
  ratio: GROUP_CONTEXT_PRESSURE_RATIO,
455
- turnCount,
456
441
  usedFallbackContextWindow: !hasModelsDevContext && !hasConfigContext && contextWindow === DEFAULT_CONTEXT_WINDOW,
457
442
  };
458
443
  }
@@ -464,13 +449,12 @@ export function shouldAllowGroupReflection({
464
449
  * @typedef {{ type: 'turn_end', turnNumber: number, stopReason: string, terminal?: boolean }} TurnEndEvent
465
450
  * @typedef {{ type: 'tool_start', id: string, name: string, input: object }} ToolStartEvent
466
451
  * @typedef {{ type: 'tool_end', id: string, name: string, output: string, isError: boolean, skipped?: boolean }} ToolEndEvent
467
- * @typedef {{ type: 'consolidate', archivedCount: number, extractedCount: number }} ConsolidateEvent
468
452
  * @typedef {{ type: 'recall', entryCount: number, cached: boolean }} RecallEvent
469
453
  * @typedef {{ type: 'fallback', from: string, to: string, reason: string }} FallbackEvent
470
454
  * @typedef {{ type: 'llm_retry', attempt: number, maxRetries: number, delayMs: number, reason: 'rate_limit_retry_after'|'rate_limit_backoff'|'transient_backoff'|'stream_idle_timeout', recoveryMode: 'restart'|'continue', errorName: string, statusCode: number|null, message: string }} LlmRetryEvent
471
455
  * @typedef {{ type: 'error', error: Error, retryable: boolean, reason?: 'stream_idle_timeout', retryExhausted?: boolean, retryAttempts?: number, maxRetries?: number }} ErrorEvent
472
456
  *
473
- * @typedef {import('./llm/adapter.js').StreamEvent | TurnStartEvent | TurnEndEvent | ToolStartEvent | ToolEndEvent | ConsolidateEvent | RecallEvent | FallbackEvent | LlmRetryEvent | ErrorEvent} EngineEvent
457
+ * @typedef {import('./llm/adapter.js').StreamEvent | TurnStartEvent | TurnEndEvent | ToolStartEvent | ToolEndEvent | RecallEvent | FallbackEvent | LlmRetryEvent | ErrorEvent} EngineEvent
474
458
  */
475
459
 
476
460
  // ─── Engine ──────────────────────────────────────────────────────
@@ -691,9 +675,6 @@ export class Engine {
691
675
  /** @type {import('./stats/tool-usage.js').ToolUsageStats|null} — per-tool call/latency counters */
692
676
  #toolStats = null;
693
677
 
694
- /** @type {object|null} — Config override for internal compact/recall tasks using fastModel; Dream uses the Session primary model */
695
- #fastConfig;
696
-
697
678
  /** @type {((agentId: string, evt: object) => void) | null} */
698
679
  #subAgentEventSink = null;
699
680
 
@@ -903,11 +884,8 @@ export class Engine {
903
884
  this.#managedCliReady = managedCliReady || null;
904
885
  this.#toolStats = toolStats || null;
905
886
  // Per-VP fan-out (2026-06-01): engine instances in the group path are
906
- // keyed by ${sessionId}::${vpId}::${threadId}, so binding the engine to
907
- // its (sessionId, vpId) pair at construction lets post-turn compact
908
- // scope its read/write to THIS VP's view of the conversation instead
909
- // of clobbering a session-global compact.md. Legacy / sub-agent
910
- // callers leave both null → fall back to the global file.
887
+ // keyed by ${sessionId}::${vpId}::${threadId}, so bind the engine to its
888
+ // session/VP identity for history and memory ownership.
911
889
  this.#sessionId = (typeof sessionId === 'string' && sessionId) ? sessionId : null;
912
890
  this.#vpId = (typeof vpId === 'string' && vpId) ? vpId : null;
913
891
  this.#chatId = (typeof chatId === 'string' && chatId) ? chatId : null;
@@ -1001,10 +979,6 @@ export class Engine {
1001
979
  this.#skillManager = this.#baseSkillManager && Array.isArray(config.plugins?.skills)
1002
980
  ? createPluginSkillManager(this.#baseSkillManager, config.plugins)
1003
981
  : this.#baseSkillManager;
1004
- const fastModelId = config.fastModelId || config.model;
1005
- this.#fastConfig = fastModelId !== config.model
1006
- ? { ...config, model: fastModelId }
1007
- : config;
1008
982
  }
1009
983
 
1010
984
  /**
@@ -1585,28 +1559,6 @@ export class Engine {
1585
1559
  return memory;
1586
1560
  }
1587
1561
 
1588
- /**
1589
- * Read compact summary from conversation store.
1590
- *
1591
- * @returns {string}
1592
- */
1593
- #getCompactSummary() {
1594
- if (!this.#conversationStore) return '';
1595
- // Per-(group, vp) scoping: when this engine is bound to a fan-out VP,
1596
- // read ONLY its own summary file. Falling back to legacy compact.md here
1597
- // leaks another group/VP's summary into every new group turn after one
1598
- // post-turn compact writes the session-global file.
1599
- if (this.#chatId && this.#vpId
1600
- && typeof this.#conversationStore.readCompactSummaryForChat === 'function') {
1601
- return this.#conversationStore.readCompactSummaryForChat(this.#chatId, this.#vpId);
1602
- }
1603
- if (this.#sessionId && this.#vpId
1604
- && typeof this.#conversationStore.readCompactSummaryFor === 'function') {
1605
- return this.#conversationStore.readCompactSummaryFor(this.#sessionId, this.#vpId);
1606
- }
1607
- return this.#conversationStore.readCompactSummary();
1608
- }
1609
-
1610
1562
  #canPersistConversation() {
1611
1563
  return Boolean(this.#conversationStore) && !this.#config._readOnly;
1612
1564
  }
@@ -1685,214 +1637,6 @@ export class Engine {
1685
1637
  return this.#conversationStore.foldMessages(persistedRows, record);
1686
1638
  }
1687
1639
 
1688
- /**
1689
- * Check and trigger consolidation if needed.
1690
- * Skipped in read-only mode.
1691
- *
1692
- * @returns {Promise<{ archivedCount: number, extractedCount: number }|null>}
1693
- */
1694
- async #maybeConsolidate() {
1695
- if (!this.#conversationStore) return null;
1696
- if (this.#config._readOnly) return null;
1697
-
1698
- const budget = this.#config.messageTokenBudget || 32768;
1699
- const compactCfg = (this.#config && this.#config.compact) || {};
1700
- return this.#runOrchestratorCompact(budget, compactCfg);
1701
- }
1702
-
1703
- /**
1704
- * Run compact via the orchestrator (DESIGN §4.2).
1705
- *
1706
- * PR-B rip: the legacy entries-based extract hook is gone — Dream V2
1707
- * owns durable memory extraction now. The orchestrator runs Track 1
1708
- * (compaction + summary) and Track 2 (task summary refresh, when wired);
1709
- * Track 3 (extract) is intentionally omitted.
1710
- *
1711
- * @param {number} budget
1712
- * @param {object} compactCfg
1713
- * @returns {Promise<{archivedCount:number, extractedCount:number}|null>}
1714
- */
1715
- async #runOrchestratorCompact(budget, _compactCfg) {
1716
- const conversationStore = this.#conversationStore;
1717
- const adapter = this.#adapter;
1718
- const fastConfig = this.#fastConfig;
1719
-
1720
- // Per-(group, vp) scoping: when this engine is bound to a fan-out VP
1721
- // (the common case in group mode), load only the rows THIS VP saw in
1722
- // its context — user prompts + every VP's assistant text, with other
1723
- // VPs' tool calls/results stripped (see persist.loadSessionHistoryForVp).
1724
- //
1725
- // Legacy / sub-agent callers (no sessionId/vpId pair) keep the global
1726
- // loadAll() behaviour so we don't break those flows.
1727
- let messages;
1728
- const scopedChat = !!(this.#chatId && this.#vpId
1729
- && typeof conversationStore.loadChatHistoryForVp === 'function');
1730
- const scoped = !scopedChat && !!(this.#sessionId && this.#vpId
1731
- && typeof conversationStore.loadSessionHistoryForVp === 'function');
1732
- try {
1733
- messages = scopedChat
1734
- ? conversationStore.loadChatHistoryForVp(this.#chatId, this.#vpId)
1735
- : scoped
1736
- ? conversationStore.loadSessionHistoryForVp(this.#sessionId, this.#vpId)
1737
- : conversationStore.loadAll();
1738
- } catch { return null; }
1739
- if (!Array.isArray(messages) || messages.length === 0) return null;
1740
-
1741
- const tokenCount = conversationStore.hotTokens();
1742
- // In the scoped path, sessionId is the engine's binding (authoritative).
1743
- // In the legacy path, fall back to scanning the messages (best-effort,
1744
- // used only for the group context-window gate).
1745
- const sessionId = this.#sessionId
1746
- || messages.find(m => m && typeof m.sessionId === 'string' && m.sessionId)?.sessionId
1747
- || null;
1748
- const groupContextGate = shouldAllowGroupReflection({
1749
- system: '',
1750
- messages,
1751
- model: this.#config.model,
1752
- config: this.#config,
1753
- sessionId,
1754
- });
1755
- if (sessionId && groupContextGate?.usedFallbackContextWindow) {
1756
- this.#trace.log?.('group_context_window_fallback', {
1757
- sessionId,
1758
- model: this.#config.model,
1759
- contextWindow: groupContextGate.contextWindow,
1760
- threshold: groupContextGate.threshold,
1761
- });
1762
- }
1763
- if (sessionId && !groupContextGate.compactAllowed) return null;
1764
-
1765
- const trig = evaluateCompactTriggers({
1766
- messages,
1767
- tokenCount,
1768
- contextLimit: this.#config.maxContextTokens || 200000,
1769
- tokenRatio: sessionId ? GROUP_CONTEXT_PRESSURE_RATIO : undefined,
1770
- maxMessages: sessionId ? Number.POSITIVE_INFINITY : undefined,
1771
- });
1772
- if (!trig.trigger) return null;
1773
-
1774
- // Use partitionMessages to decide what is "cooling": orchestrator's
1775
- // own keepHot is a count, but we want to honour the token-budget
1776
- // partitioning the rest of the system uses.
1777
- const { toArchive } = partitionMessages(messages, budget);
1778
- if (toArchive.length === 0) return null;
1779
-
1780
- const archiveIds = [];
1781
-
1782
- // Language-aware summarizer prompts. The orchestrator-track summary
1783
- // ends up in the system prompt as a "previous conversation summary"
1784
- // block, so it needs to match the user's preferred language to avoid
1785
- // a jarring locale flip mid-context.
1786
- const isZh = String(this.#config.language || '').toLowerCase().startsWith('zh');
1787
- const summariserSystem = isZh
1788
- ? '你是对话摘要器。下面包含「先前累计摘要」(可能为空)与「新待压缩对话」。请融合两者,输出一份「重写后的累计摘要」——不要分段罗列日期、不要保留 "## 2026-..." 等历史分节,直接产出一份连贯、可被下一轮直接重新注入 prompt 的摘要。保留关键决策、事实、上下文与人物意图。'
1789
- : 'You are a conversation summarizer. The input contains a "previous cumulative summary" (may be empty) plus a "new conversation to absorb". Merge them into ONE rewritten cumulative summary — do NOT keep dated section headers or any historical log structure. Output a single coherent summary suitable to be re-injected into the next turn\'s prompt as-is. Preserve key decisions, facts, context, and intent.';
1790
- const summariserPromptPrefix = isZh ? '请概括:\n\n' : 'Summarize:\n\n';
1791
-
1792
- const hooks = {
1793
- summarise: async () => {
1794
- try {
1795
- // Read prior summary at call time, not at orchestrator setup,
1796
- // so the merge always sees the freshest on-disk state even if
1797
- // future orchestrator changes invoke summarise more than once.
1798
- const priorSummary = this.#getCompactSummary() || '';
1799
- const priorBlock = priorSummary
1800
- ? (isZh
1801
- ? `【先前累计摘要】\n${priorSummary}\n\n【新待压缩对话】\n`
1802
- : `[Previous cumulative summary]\n${priorSummary}\n\n[New conversation to absorb]\n`)
1803
- : '';
1804
- const maintenanceCtrl = new AbortController();
1805
- let timeout = null;
1806
- const timedOut = new Promise((_, reject) => {
1807
- timeout = setTimeout(() => {
1808
- maintenanceCtrl.abort('compact_summary_timeout');
1809
- reject(new LLMAbortError());
1810
- }, MAINTENANCE_CALL_TIMEOUT_MS);
1811
- if (timeout && typeof timeout.unref === 'function') timeout.unref();
1812
- });
1813
- try {
1814
- const request = adapter.call({
1815
- model: fastConfig.model,
1816
- system: summariserSystem,
1817
- messages: [{ role: 'user', content: `${summariserPromptPrefix}${priorBlock}${toArchive.map(m => `[${m.role}] ${(m.content || '').slice(0, 500)}`).join('\n\n')}` }],
1818
- // 10k output budget: the running summary is the engine's
1819
- // long-term memory of cold turns, so it deserves room to
1820
- // actually preserve detail. We rewrite-in-place each round,
1821
- // so size stays bounded by maxTokens regardless of how many
1822
- // compact passes have run.
1823
- maxTokens: 10240,
1824
- signal: maintenanceCtrl.signal,
1825
- });
1826
- const result = await Promise.race([request, timedOut]);
1827
- return (result.text || '').trim();
1828
- } finally {
1829
- if (timeout) clearTimeout(timeout);
1830
- }
1831
- } catch {
1832
- return '';
1833
- }
1834
- },
1835
- archive: async (_groupIdx, groupMsgs) => {
1836
- // Only collect archive ids when we'll actually use them. In the
1837
- // scoped (per-VP) path we never call moveToColdBatch — those
1838
- // rows are shared with sibling VPs in this group — so leaving
1839
- // the push in would be dead state a future reader has to chase.
1840
- if (!scoped) {
1841
- for (const m of groupMsgs) if (m.id) archiveIds.push(m.id);
1842
- }
1843
- const turnId = groupMsgs[0]?.id || `g_${Date.now()}`;
1844
- if (this.#yeaftDir) {
1845
- try {
1846
- await archiveTurn({
1847
- root: `${this.#yeaftDir}/memory`,
1848
- scopeDir: 'user',
1849
- turnId,
1850
- messages: groupMsgs,
1851
- });
1852
- } catch { /* best-effort */ }
1853
- }
1854
- return { turnId };
1855
- },
1856
- };
1857
-
1858
- try {
1859
- const out = await runCompactOrchestrator({
1860
- messages, keepHot: 10, hooks,
1861
- });
1862
- // Scoped path (per-(group, vp)): do NOT moveToColdBatch — those
1863
- // archive ids include user rows and other VPs' assistant rows that
1864
- // sibling VPs in this group still need in their hot context. The
1865
- // per-VP summary written below is the durable win; physical
1866
- // cold-archival across shared rows is the dream-level orchestrator's
1867
- // job, not post-turn compact's.
1868
- if (!scoped && !scopedChat && archiveIds.length > 0) {
1869
- conversationStore.moveToColdBatch(archiveIds);
1870
- }
1871
- if (out.compactSummary) {
1872
- if (scopedChat && typeof conversationStore.replaceCompactSummaryForChat === 'function') {
1873
- conversationStore.replaceCompactSummaryForChat(this.#chatId, this.#vpId, out.compactSummary);
1874
- } else if (scoped && typeof conversationStore.replaceCompactSummaryFor === 'function') {
1875
- conversationStore.replaceCompactSummaryFor(this.#sessionId, this.#vpId, out.compactSummary);
1876
- } else {
1877
- conversationStore.replaceCompactSummary(out.compactSummary);
1878
- }
1879
- }
1880
- // Index update only makes sense for the legacy path that actually
1881
- // moved rows to cold. In the scoped path, nothing on disk changed.
1882
- if (!scoped) {
1883
- const lastKept = messages[messages.length - 1];
1884
- conversationStore.updateIndex({ lastMessageId: lastKept?.id || null });
1885
- }
1886
-
1887
- return {
1888
- archivedCount: out.archivedMessages,
1889
- extractedCount: out.extractedCount,
1890
- };
1891
- } catch {
1892
- return null;
1893
- }
1894
- }
1895
-
1896
1640
  #formatTaskResultUpdateContent(content) {
1897
1641
  if (typeof content === 'string') return content;
1898
1642
  try { return JSON.stringify(content); } catch { return String(content); }
@@ -2569,55 +2313,7 @@ export class Engine {
2569
2313
  });
2570
2314
  let systemPrompt = buildCurrentSystemPrompt();
2571
2315
 
2572
- // ─── HARD INVARIANT: Compact ≠ Dream (read DESIGN-COMPACT-VS-DREAM.md) ─
2573
- // Compact summary (this block) ONLY lands in the messages array head as
2574
- // a `<conversation_summary>` user/assistant pair. It MUST NEVER appear
2575
- // in the system prompt — that was the bug DESIGN-PROMPT §4.3 banned.
2576
- //
2577
- // Inversely: Dream's prompt-facing `content.md` flows exclusively through
2578
- // `prompts.js#buildSystemPrompt`'s Memory section via AMS Resident (see
2579
- // `engine.js#buildResidentEntries`). Evidence and catalog files stay out.
2580
- // It MUST NEVER appear in the messages array.
2581
- //
2582
- // Two write roots, two scheduler triggers, two prompt slots — never
2583
- // mixed. Anyone touching this section must read
2584
- // `agent/yeaft/DESIGN-COMPACT-VS-DREAM.md` before changing the wiring;
2585
- // the boundary has been violated twice in this codebase's history and
2586
- // each time it took an LLM cache-thrash + persona-dup follow-up PR to
2587
- // unwind.
2588
- //
2589
- // ─── Compact summary as messages-array head (DESIGN-PROMPT §4.3) ─
2590
- // The previous code placed the compact summary inside the system
2591
- // prompt; that broke prompt-cache hit-rate (any compact update
2592
- // invalidated the entire system) and conflated identity/rules with
2593
- // dialogue history. The compact summary is the product of compressing
2594
- // older turns, so it belongs at the head of the messages array.
2595
- //
2596
- // Note: this is a separate mechanism from `history-compact.js`'s
2597
- // `_compactSummary`-tagged user message. They never collide:
2598
- // • THIS path injects a `<conversation_summary>` pair on every
2599
- // query when conversationStore.readCompactSummary() returns text
2600
- // (i.e. when a previous T1 run wrote one to disk). Engine reads,
2601
- // does not produce.
2602
- // • history-compact.js#compactHistory rewrites the in-memory
2603
- // `messages` array, replacing cold messages with a single
2604
- // `_compactSummary`-tagged user message. That path runs at a
2605
- // different layer (web-bridge during a manual /compact) and never
2606
- // touches `compactMessages` here.
2607
- // The two would only overlap if a tagged `_compactSummary` user
2608
- // message also matched the `<conversation_summary>` template — they
2609
- // don't, so duplication is impossible by construction.
2610
- const compactSummaryRaw = this.#getCompactSummary();
2611
- const compactSummary = typeof compactSummaryRaw === 'string'
2612
- ? compactSummaryRaw.trim() : '';
2613
- const compactMessages = compactSummary
2614
- ? [
2615
- { role: 'user', content: `<conversation_summary>\n${compactSummary}\n</conversation_summary>` },
2616
- { role: 'assistant', content: 'Acknowledged.' },
2617
- ]
2618
- : [];
2619
-
2620
- // Build conversation: optional compact head + existing messages + new user message.
2316
+ // Build conversation from the caller-provided history and the new user message.
2621
2317
  // If `promptParts` was supplied (image/file attachments), use the array form
2622
2318
  // so the adapter sees image content blocks alongside the text. Otherwise the
2623
2319
  // legacy string form keeps prompt-cache behavior identical.
@@ -2653,8 +2349,10 @@ export class Engine {
2653
2349
  : prompt;
2654
2350
  }
2655
2351
  const conversationMessages = [
2656
- ...compactMessages,
2657
- ...messages,
2352
+ ...trimSnapshotForBudget(messages, {
2353
+ messageTokenBudget: this.#config.messageTokenBudget,
2354
+ language: this.#config.language,
2355
+ }),
2658
2356
  { role: 'user', content: finalUserContent },
2659
2357
  ];
2660
2358
 
@@ -2805,8 +2503,8 @@ export class Engine {
2805
2503
  // on any successful stream() iteration, and also on a fallback-model
2806
2504
  // switch (the new model gets a fresh budget). Reaching maxRetries
2807
2505
  // gives up: we either fall back to a backup model or surface the
2808
- // error to the user. LLMContextError has its own compact-retry path
2809
- // and does NOT count against this budget.
2506
+ // error to the user. Context errors are surfaced immediately and do
2507
+ // not count against this budget.
2810
2508
  let retryPolicy = resolveRetryPolicy(this.#config);
2811
2509
  let consecutiveRetryableErrors = 0;
2812
2510
  let consecutiveForbiddenErrors = 0;
@@ -2842,8 +2540,7 @@ export class Engine {
2842
2540
  retryPolicy = resolveRetryPolicy(requestConfig);
2843
2541
 
2844
2542
  // task-324: no hard MAX_TURNS cap. Loop terminates on end_turn,
2845
- // non-retryable error, LLMContextError (after compact retry), or
2846
- // caller abort. Keeping this comment so the removal is traceable.
2543
+ // non-retryable error, LLMContextError, or caller abort.
2847
2544
 
2848
2545
  // task-325a: check for user abort at the top of every turn so a
2849
2546
  // signal that fires between turns (e.g. during tool execution in
@@ -3065,9 +2762,17 @@ export class Engine {
3065
2762
  // message_trace can fetch it on demand. The stub keeps the
3066
2763
  // OpenAI/Anthropic toolCallId pairing intact.
3067
2764
  const pendingContinuationForRequest = retryLifecycle.pendingContinuation;
2765
+ // Bound only this provider copy. The durable transcript and the live
2766
+ // query tape remain complete; no summary is generated and no history
2767
+ // rows are rewritten. This also protects later tool-loop requests,
2768
+ // not just the initial snapshot assembled by the bridge.
2769
+ const requestHistory = trimSnapshotForBudget(conversationMessages, {
2770
+ messageTokenBudget: requestConfig.messageTokenBudget,
2771
+ language: requestConfig.language,
2772
+ });
3068
2773
  let wireMessages = stripMetaForWire(pendingContinuationForRequest
3069
- ? [...conversationMessages, pendingContinuationForRequest]
3070
- : [...conversationMessages]);
2774
+ ? [...requestHistory, pendingContinuationForRequest]
2775
+ : requestHistory);
3071
2776
 
3072
2777
  if (scenario !== 'work-item' && this.#yeaftDir && (this.#config?.archive?.toolResults !== false)) {
3073
2778
  try {
@@ -3078,15 +2783,10 @@ export class Engine {
3078
2783
  turnAgeMin: this.#config?.archive?.turnAgeMin,
3079
2784
  lengthMin: this.#config?.archive?.lengthMin,
3080
2785
  });
2786
+ // Keep the live query tape and persisted transcript raw. The
2787
+ // archive result is a provider-only copy; a later request may
2788
+ // repeat this best-effort archive lookup without losing history.
3081
2789
  wireMessages = swept.nextMessages;
3082
- // Mutate the in-memory conversation array so subsequent turns
3083
- // see the stub too — without this, the next turn re-archives
3084
- // the same body.
3085
- if (swept.archivedCount > 0) {
3086
- for (let i = 0; i < conversationMessages.length; i += 1) {
3087
- conversationMessages[i] = wireMessages[i];
3088
- }
3089
- }
3090
2790
  } catch { /* best-effort */ }
3091
2791
  }
3092
2792
 
@@ -3121,9 +2821,6 @@ export class Engine {
3121
2821
  });
3122
2822
  wireMessages = sweep.nextMessages;
3123
2823
  if (sweep.archivedCount > 0) {
3124
- for (let i = 0; i < conversationMessages.length; i += 1) {
3125
- conversationMessages[i] = wireMessages[i];
3126
- }
3127
2824
  this.#trace.log?.('preflight_sweep', {
3128
2825
  archivedCount: sweep.archivedCount,
3129
2826
  archivedBytes: sweep.archivedBytes,
@@ -3445,16 +3142,6 @@ export class Engine {
3445
3142
  break;
3446
3143
  }
3447
3144
 
3448
- if (err instanceof LLMContextError && this.#conversationStore) {
3449
- const retryConsolidated = await this.#maybeConsolidate();
3450
- if (retryConsolidated && retryConsolidated.archivedCount > 0) {
3451
- endAttemptTrace('context_overflow_retry');
3452
- yield { type: 'consolidate', archivedCount: retryConsolidated.archivedCount, extractedCount: retryConsolidated.extractedCount };
3453
- yield { type: 'turn_end', turnNumber, stopReason: 'context_overflow_retry', threadId };
3454
- continue;
3455
- }
3456
- }
3457
-
3458
3145
  traceRequest('llm.request_error', {
3459
3146
  durationMs: perfNowMs() - requestPerfStart,
3460
3147
  ok: false,
@@ -4042,12 +3729,7 @@ export class Engine {
4042
3729
  yield { type: 'turn_end', turnNumber, stopReason, threadId, terminal: true };
4043
3730
 
4044
3731
  // Message durability is handled incrementally before this terminal
4045
- // branch. End-of-turn owns maintenance only; re-appending the whole
4046
- // turn here would duplicate rows and reintroduce the crash window.
4047
- const consolidated = await this.#maybeConsolidate();
4048
- if (consolidated && consolidated.archivedCount > 0) {
4049
- yield { type: 'consolidate', archivedCount: consolidated.archivedCount, extractedCount: consolidated.extractedCount };
4050
- }
3732
+ // branch. There is no hidden end-of-turn LLM maintenance step.
4051
3733
 
4052
3734
  // PR-L: T2 end-of-turn (asynchronous) reflection. Fires when the
4053
3735
  // total tool count for this query() exceeds TURN_SUMMARY_THRESHOLD
@@ -5437,33 +5119,6 @@ export class Engine {
5437
5119
  /** @returns {string|null} */
5438
5120
  get yeaftDir() { return this.#yeaftDir; }
5439
5121
 
5440
- /** @returns {object} — Config with fastModel as model (for compact and other non-Dream internal tasks) */
5441
- get fastConfig() { return this.#fastConfig; }
5442
-
5443
- /**
5444
- * Run a one-shot fast-model call to produce a compact summary.
5445
- * Used by the web bridge's in-memory history compactor
5446
- * (`agent/yeaft/history-compact.js`) — kept on the engine so callers
5447
- * don't reach into the private adapter field.
5448
- *
5449
- * @param {{system: string, prompt: string, maxTokens?: number}} args
5450
- * @returns {Promise<string>} — summary text (trimmed); '' on failure
5451
- */
5452
- async summarizeForCompact({ system, prompt, maxTokens = 1024 } = {}) {
5453
- if (!system || !prompt) return '';
5454
- try {
5455
- const out = await this.#adapter.call({
5456
- model: this.#fastConfig.model,
5457
- system,
5458
- messages: [{ role: 'user', content: prompt }],
5459
- maxTokens,
5460
- });
5461
- return (out?.text || '').trim();
5462
- } catch (err) {
5463
- console.warn('[Engine] summarizeForCompact failed:', err?.message || err);
5464
- return '';
5465
- }
5466
- }
5467
5122
  }
5468
5123
 
5469
5124
  function selectExactSessionScope(selected, sessionId) {