@yeaft/webchat-agent 1.0.503 → 1.0.505

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/local-runtime/server/db/connection.js +14 -2
  2. package/local-runtime/server/db/user-stats-db.js +7 -0
  3. package/local-runtime/server/handlers/agent-conversation.js +2 -1
  4. package/local-runtime/server/handlers/agent-output.js +4 -1
  5. package/local-runtime/server/routes/admin-routes.js +3 -3
  6. package/local-runtime/version.json +1 -1
  7. package/local-runtime/web/app.bundle.js +108 -120
  8. package/local-runtime/web/app.bundle.js.gz +0 -0
  9. package/local-runtime/web/index.html +4 -3
  10. package/local-runtime/web/style.bundle.css +1 -1
  11. package/local-runtime/web/style.bundle.css.gz +0 -0
  12. package/local-runtime/web/vendor/katex/LICENSE +21 -0
  13. package/local-runtime/web/vendor/katex/fonts/KaTeX_AMS-Regular.ttf +0 -0
  14. package/local-runtime/web/vendor/katex/fonts/KaTeX_AMS-Regular.woff +0 -0
  15. package/local-runtime/web/vendor/katex/fonts/KaTeX_AMS-Regular.woff2 +0 -0
  16. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Bold.ttf +0 -0
  17. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Bold.woff +0 -0
  18. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Bold.woff2 +0 -0
  19. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Regular.ttf +0 -0
  20. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Regular.woff +0 -0
  21. package/local-runtime/web/vendor/katex/fonts/KaTeX_Caligraphic-Regular.woff2 +0 -0
  22. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Bold.ttf +0 -0
  23. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Bold.woff +0 -0
  24. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Bold.woff2 +0 -0
  25. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Regular.ttf +0 -0
  26. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Regular.woff +0 -0
  27. package/local-runtime/web/vendor/katex/fonts/KaTeX_Fraktur-Regular.woff2 +0 -0
  28. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Bold.ttf +0 -0
  29. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Bold.woff +0 -0
  30. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Bold.woff2 +0 -0
  31. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-BoldItalic.ttf +0 -0
  32. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-BoldItalic.woff +0 -0
  33. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-BoldItalic.woff2 +0 -0
  34. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Italic.ttf +0 -0
  35. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Italic.woff +0 -0
  36. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Italic.woff2 +0 -0
  37. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Regular.ttf +0 -0
  38. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Regular.woff +0 -0
  39. package/local-runtime/web/vendor/katex/fonts/KaTeX_Main-Regular.woff2 +0 -0
  40. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-BoldItalic.ttf +0 -0
  41. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-BoldItalic.woff +0 -0
  42. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-BoldItalic.woff2 +0 -0
  43. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-Italic.ttf +0 -0
  44. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-Italic.woff +0 -0
  45. package/local-runtime/web/vendor/katex/fonts/KaTeX_Math-Italic.woff2 +0 -0
  46. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Bold.ttf +0 -0
  47. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Bold.woff +0 -0
  48. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Bold.woff2 +0 -0
  49. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Italic.ttf +0 -0
  50. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Italic.woff +0 -0
  51. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Italic.woff2 +0 -0
  52. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Regular.ttf +0 -0
  53. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Regular.woff +0 -0
  54. package/local-runtime/web/vendor/katex/fonts/KaTeX_SansSerif-Regular.woff2 +0 -0
  55. package/local-runtime/web/vendor/katex/fonts/KaTeX_Script-Regular.ttf +0 -0
  56. package/local-runtime/web/vendor/katex/fonts/KaTeX_Script-Regular.woff +0 -0
  57. package/local-runtime/web/vendor/katex/fonts/KaTeX_Script-Regular.woff2 +0 -0
  58. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size1-Regular.ttf +0 -0
  59. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size1-Regular.woff +0 -0
  60. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size1-Regular.woff2 +0 -0
  61. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size2-Regular.ttf +0 -0
  62. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size2-Regular.woff +0 -0
  63. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size2-Regular.woff2 +0 -0
  64. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size3-Regular.ttf +0 -0
  65. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size3-Regular.woff +0 -0
  66. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size3-Regular.woff2 +0 -0
  67. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size4-Regular.ttf +0 -0
  68. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size4-Regular.woff +0 -0
  69. package/local-runtime/web/vendor/katex/fonts/KaTeX_Size4-Regular.woff2 +0 -0
  70. package/local-runtime/web/vendor/katex/fonts/KaTeX_Typewriter-Regular.ttf +0 -0
  71. package/local-runtime/web/vendor/katex/fonts/KaTeX_Typewriter-Regular.woff +0 -0
  72. package/local-runtime/web/vendor/katex/fonts/KaTeX_Typewriter-Regular.woff2 +0 -0
  73. package/local-runtime/web/vendor/katex/katex.min.css +1 -0
  74. package/local-runtime/web/vendor.bundle.js +1 -0
  75. package/local-runtime/web/vendor.bundle.js.gz +0 -0
  76. package/package.json +1 -1
  77. package/yeaft/config-api.js +16 -5
  78. package/yeaft/config.js +11 -2
  79. package/yeaft/conversation/history-index-worker.js +151 -0
  80. package/yeaft/conversation/history-index.js +96 -1
  81. package/yeaft/conversation/persist.js +6 -1
  82. package/yeaft/conversation/recall-relevance.js +103 -0
  83. package/yeaft/debug-trace.js +15 -4
  84. package/yeaft/engine.js +89 -16
  85. package/yeaft/history-window.js +370 -0
  86. package/yeaft/vp/seed-defaults.js +169 -8
  87. package/yeaft/vp/stock-ids.js +2 -0
  88. package/yeaft/web-bridge.js +19 -12
package/yeaft/engine.js CHANGED
@@ -33,7 +33,9 @@ import {
33
33
  DEFAULT_PROJECT_DOC_MAX_BYTES,
34
34
  } from './sessions/project-doc.js';
35
35
  import { archiveToolResults } from './archive/tool-results.js';
36
- import { trimSnapshotForBudget } from './history-window.js';
36
+ import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets } from './history-window.js';
37
+ import { recallConversationTurns } from './conversation/history-index.js';
38
+ import { parseSeqFromId } from './conversation/persist.js';
37
39
  import { isVpForeign, readContent as readScopeContent } from './memory/store.js';
38
40
  import { ActiveMemorySet } from './memory/ams.js';
39
41
  import { cleanMemoryPromptText } from './memory/prompt-cleanup.js';
@@ -1875,7 +1877,7 @@ export class Engine {
1875
1877
  }
1876
1878
  }
1877
1879
 
1878
- async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1880
+ async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1879
1881
  if (!prompt || typeof prompt !== 'string' || !prompt.trim()) {
1880
1882
  const error = new Error('prompt is required and must be a non-empty string');
1881
1883
  yield {
@@ -1967,7 +1969,7 @@ export class Engine {
1967
1969
  try {
1968
1970
  this.#currentThreadId = threadId || MAIN_THREAD_ID;
1969
1971
  this.#currentCausalRootId = effectiveCausalRootId;
1970
- yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1972
+ yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1971
1973
  } finally {
1972
1974
  // Closing the async generator at a visible retry boundary means the
1973
1975
  // continuation never reached a provider. Keep it out of history and
@@ -2022,7 +2024,7 @@ export class Engine {
2022
2024
  * in a try/finally without indenting the whole loop.
2023
2025
  * @private
2024
2026
  */
2025
- async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2027
+ async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2026
2028
 
2027
2029
  const effectiveCollabToolPolicy = collabToolPolicy === COLLAB_TOOL_POLICY.SINGLE_VP || collabToolPolicy === COLLAB_TOOL_POLICY.MULTI_VP
2028
2030
  ? collabToolPolicy
@@ -2079,10 +2081,59 @@ export class Engine {
2079
2081
  // memory pre-flow or provider request can fail. The Web Session bridge
2080
2082
  // already writes one shared user row before multi-VP fan-out, so those
2081
2083
  // callers set userAlreadyPersisted and every VP skips this append.
2084
+ const internalTrigger = !!inboundEnvelope?.msg?.meta?.injectedBy;
2085
+ let persistedQueryUser = currentUserMessage;
2082
2086
  if (!userAlreadyPersisted) {
2083
- this.#persistConversationMessage({ role: 'user', content: prompt, userAuthored: true }, {
2084
- sessionId: runtimeSessionId,
2085
- });
2087
+ persistedQueryUser = this.#persistConversationMessage({
2088
+ role: internalTrigger ? 'assistant' : 'user', content: prompt,
2089
+ ...(internalTrigger ? { internal: true } : { userAuthored: true }),
2090
+ }, { sessionId: runtimeSessionId });
2091
+ }
2092
+ // Native Session history, including internal wakeups, never comes from
2093
+ // Dream. Work Center and scoped child agents keep their memory contracts.
2094
+ const useMessageHistory = scenario !== 'work-item' && !!runtimeSessionId && !vpPersona?.subAgent;
2095
+ const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
2096
+ || (!runtimeSessionId && !internalTrigger);
2097
+ const recentTurnCap = this.#config.yeaft?.recentTurnsLimit ?? 20;
2098
+ const relatedTurnCap = this.#config.yeaft?.relatedTurnsLimit ?? 8;
2099
+ let relatedHistoryTurns = [];
2100
+ let historyRecallMeta = { source: 'messages', status: 'disabled' };
2101
+ if (useMessageHistory && !internalTrigger && this.#conversationStore?.loadRecentBySession) {
2102
+ // Bounded canonical rows, not the lossy bridge cache. Threads share the
2103
+ // Session transcript; only tool arcs and thinking are VP-private.
2104
+ const clientId = inboundEnvelope?.msg?.meta?.clientMessageId;
2105
+ if (!persistedQueryUser && clientId) {
2106
+ // Compatibility only: bridge callers carry the durable row directly.
2107
+ // A missing identity outside this bounded lookup fails closed below.
2108
+ persistedQueryUser = this.#conversationStore.loadRecentBySession(runtimeSessionId, recentTurnCap + 1)
2109
+ .find(m => m.role === 'user' && m.clientMessageId === clientId);
2110
+ }
2111
+ const beforeSeq = Number.isFinite(persistedQueryUser?.seq)
2112
+ ? persistedQueryUser.seq : parseSeqFromId(persistedQueryUser?.id);
2113
+ if (Number.isFinite(beforeSeq)) {
2114
+ const tail = this.#conversationStore.loadRecentBySession(runtimeSessionId, recentTurnCap, { beforeSeq });
2115
+ messages = tail.filter(m => parseSeqFromId(m.id) < beforeSeq
2116
+ && (m.role !== 'tool' || !queryVpId || m.speakerVpId === queryVpId))
2117
+ .map(m => {
2118
+ if (m.role !== 'assistant' || !queryVpId || m.speakerVpId === queryVpId) return m;
2119
+ const { toolCalls, thinkingBlocks, ...textOnly } = m;
2120
+ return textOnly;
2121
+ });
2122
+ if (relatedTurnCap > 0 && this.#yeaftDir) {
2123
+ try {
2124
+ const recalled = await recallConversationTurns(this.#yeaftDir, runtimeSessionId, prompt, {
2125
+ beforeSeq, limit: relatedTurnCap,
2126
+ });
2127
+ relatedHistoryTurns = recalled.turns || [];
2128
+ historyRecallMeta = { source: 'messages', status: 'ready', ...recalled.meta };
2129
+ } catch (error) {
2130
+ // Cold/stale index degrades to recent history, never a full scan.
2131
+ historyRecallMeta = { source: 'messages', status: error?.code || 'unavailable' };
2132
+ }
2133
+ }
2134
+ } else {
2135
+ historyRecallMeta = { source: 'messages', status: 'missing_current_user_fence' };
2136
+ }
2086
2137
  }
2087
2138
 
2088
2139
  const perfTraceId = typeof inboundEnvelope?._perfTraceId === 'string' && inboundEnvelope._perfTraceId.trim()
@@ -2111,7 +2162,7 @@ export class Engine {
2111
2162
  let memoryInjection = '';
2112
2163
  let recallEntryCount = 0;
2113
2164
 
2114
- const topicScopesForMemory = await this.#loadSessionTopicScopes(sessionId);
2165
+ const topicScopesForMemory = !useDreamMemory ? [] : await this.#loadSessionTopicScopes(sessionId);
2115
2166
  const projectScopesForMemory = Array.isArray(projectSessionIds)
2116
2167
  ? projectSessionIds.flatMap(id => [
2117
2168
  `sessions/${id}`,
@@ -2119,7 +2170,7 @@ export class Engine {
2119
2170
  `group/${id}`,
2120
2171
  ])
2121
2172
  : [];
2122
- const recallResult = await this.#recallMemory(prompt, {
2173
+ const recallResult = !useDreamMemory ? { entries: [], meta: {} } : await this.#recallMemory(prompt, {
2123
2174
  sessionId,
2124
2175
  vpId: vpPersona && typeof vpPersona === 'object' && typeof vpPersona.vpId === 'string'
2125
2176
  ? vpPersona.vpId
@@ -2148,7 +2199,7 @@ export class Engine {
2148
2199
 
2149
2200
  // Load canonical content only for scopes selected by ranked FTS records.
2150
2201
  // summary.md remains catalog metadata and never enters the prompt.
2151
- const summaries = await this.#loadLayerASummaries({
2202
+ const summaries = !useDreamMemory ? {} : await this.#loadLayerASummaries({
2152
2203
  sessionId,
2153
2204
  vpId: vpPersona && typeof vpPersona === 'object' && typeof vpPersona.vpId === 'string'
2154
2205
  ? vpPersona.vpId
@@ -2167,7 +2218,7 @@ export class Engine {
2167
2218
  && typeof vpPersona.vpId === 'string'
2168
2219
  ? vpPersona.vpId
2169
2220
  : (typeof senderVpId === 'string' ? senderVpId : null);
2170
- const amsContext = this.#prepareAms({
2221
+ const amsContext = !useDreamMemory ? null : this.#prepareAms({
2171
2222
  sessionId,
2172
2223
  ownVpId: ownVpIdForAms,
2173
2224
  summaries,
@@ -2179,7 +2230,8 @@ export class Engine {
2179
2230
  }
2180
2231
  const loadedMemoryForDebug = loadedMemoryDebugEntries(amsContext?.snapshot);
2181
2232
  const loadedMemoryMetaForDebug = {
2182
- recallLimit: resolveMemoryRecallLimit(this.#config),
2233
+ ...(useMessageHistory ? historyRecallMeta : {}),
2234
+ recallLimit: useMessageHistory ? relatedTurnCap : resolveMemoryRecallLimit(this.#config),
2183
2235
  recallCandidates: Number.isFinite(recallResult?.meta?.hitCount)
2184
2236
  ? recallResult.meta.hitCount
2185
2237
  : (recallResult && Array.isArray(recallResult.entries) ? recallResult.entries.length : 0),
@@ -2356,11 +2408,14 @@ export class Engine {
2356
2408
  : prompt;
2357
2409
  }
2358
2410
  const conversationMessages = [
2359
- ...trimSnapshotForBudget(messages, {
2411
+ ...(useMessageHistory ? messages : trimSnapshotForBudget(messages, {
2360
2412
  messageTokenBudget: this.#config.messageTokenBudget,
2361
2413
  language: this.#config.language,
2362
- }),
2363
- { role: 'user', content: finalUserContent },
2414
+ })),
2415
+ { role: 'user', content: finalUserContent,
2416
+ ...(persistedQueryUser?.id ? { id: persistedQueryUser.id, seq: parseSeqFromId(persistedQueryUser.id) } : {}),
2417
+ ...(persistedQueryUser?.clientMessageId ? { clientMessageId: persistedQueryUser.clientMessageId } : {}),
2418
+ },
2364
2419
  ];
2365
2420
 
2366
2421
  const groupReflectionGate = shouldAllowGroupReflection({
@@ -2784,10 +2839,28 @@ export class Engine {
2784
2839
  // query tape remain complete; no summary is generated and no history
2785
2840
  // rows are rewritten. This also protects later tool-loop requests,
2786
2841
  // not just the initial snapshot assembled by the bridge.
2787
- const requestHistory = trimSnapshotForBudget(conversationMessages, {
2842
+ const continuationCost = pendingContinuationForRequest
2843
+ ? estimateMessageTokens(pendingContinuationForRequest) : 0;
2844
+ const historyBudget = Math.max(1, Math.min(
2845
+ requestConfig.messageTokenBudget || 32768,
2846
+ Math.floor(currentContextWindow * 0.75) - estimateMessagesTokens(systemPrompt, []),
2847
+ ) - continuationCost);
2848
+ const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
2849
+ prompt,
2850
+ relatedTurns: relatedHistoryTurns,
2851
+ recentTurnCap: requestConfig.yeaft?.recentTurnsLimit ?? 20,
2852
+ relatedTurnCap: requestConfig.yeaft?.relatedTurnsLimit ?? 8,
2853
+ messageTokenBudget: historyBudget,
2854
+ currentTurnStartIndex: turnStartIdx,
2855
+ language: requestConfig.language,
2856
+ }) : null;
2857
+ const requestHistory = buckets?.messages || trimSnapshotForBudget(conversationMessages, {
2788
2858
  messageTokenBudget: requestConfig.messageTokenBudget,
2789
2859
  language: requestConfig.language,
2790
2860
  });
2861
+ if (buckets) this.#trace.log?.('history_buckets', {
2862
+ sessionId: runtimeSessionId, turnId: queryTurnId, ...historyRecallMeta, ...buckets.meta,
2863
+ });
2791
2864
  let wireMessages = stripMetaForWire(pendingContinuationForRequest
2792
2865
  ? [...requestHistory, pendingContinuationForRequest]
2793
2866
  : requestHistory);
@@ -9,6 +9,8 @@
9
9
  */
10
10
 
11
11
  import { estimateTokens } from './conversation/persist.js';
12
+ import { isVisibleConversationRow } from './conversation/internal-control.js';
13
+ import { scoreRecallTurn } from './conversation/recall-relevance.js';
12
14
  import { pairSanitize } from './pair-sanitize.js';
13
15
  import { truncateToolResultIfNeeded } from './tools/registry.js';
14
16
  import { countTurns, indexOfNthTurnFromEnd, sliceLastNTurns } from './turn-utils.js';
@@ -693,6 +695,374 @@ export function trimSnapshotForBudget(snapshot, options = {}) {
693
695
  return withoutHistorySourceIndexes(enriched);
694
696
  }
695
697
 
698
+ // Unlike the legacy runtime-cache slicer, provider buckets never infer user
699
+ // identity from text. A repeated question is still a new human turn.
700
+ function bucketSourceIds(message) {
701
+ return [...new Set([
702
+ ...(Array.isArray(message?.sourceMessageIds) ? message.sourceMessageIds : []),
703
+ ...(Array.isArray(message?.entry?.sourceMessageIds) ? message.entry.sourceMessageIds : []),
704
+ message?._persistedMessageId, message?.id, message?.messageId,
705
+ message?.anchorMessageId,
706
+ ].filter(id => typeof id === 'string' && id))];
707
+ }
708
+
709
+ function bucketUserKeys(message) {
710
+ const keys = bucketSourceIds(message).map(id => `message:${id}`);
711
+ if (message?.clientMessageId) keys.push(`client:${message.clientMessageId}`);
712
+ if (message?.entryId) keys.push(`entry:${message.entryId}`);
713
+ return keys;
714
+ }
715
+
716
+ function bucketUserBoundary(message) {
717
+ if (message?.role !== 'user' || !isVisibleConversationRow(message)) return false;
718
+ // Anthropic's tool-result carrier is not a human turn boundary.
719
+ return !Array.isArray(message.content) || message.content.some(part => (
720
+ typeof part === 'string' || !['tool_result', 'function_call_output'].includes(part?.type)
721
+ ));
722
+ }
723
+
724
+ function bucketSequence(message) {
725
+ for (const value of [message?.userSeq, message?.entryStartSeq, message?.seq]) {
726
+ if (Number.isFinite(value)) return value;
727
+ }
728
+ for (const id of bucketSourceIds(message)) {
729
+ const match = /^m(\d+)$/.exec(id);
730
+ if (match) return Number(match[1]);
731
+ }
732
+ return null;
733
+ }
734
+
735
+ function bucketTextMessages(messages) {
736
+ return stripAllToolNoise(messages.filter(message => (
737
+ isVisibleConversationRow(message)
738
+ && (message?.role === 'user' || message?.role === 'assistant')
739
+ )));
740
+ }
741
+
742
+ function bucketTurn(messages, index, supplied = {}) {
743
+ const user = messages.find(bucketUserBoundary);
744
+ const sourceIds = [...new Set(messages.flatMap(bucketSourceIds))];
745
+ const keys = [...new Set(messages.filter(bucketUserBoundary).flatMap(bucketUserKeys))];
746
+ if (supplied.id) keys.push(`turn:${supplied.id}`);
747
+ if (keys.length === 0 && index != null) keys.push(`snapshot:${index}`);
748
+ const text = bucketTextMessages(messages);
749
+ return {
750
+ ...supplied,
751
+ id: supplied.id || keys[0] || `snapshot:${index}`,
752
+ keys,
753
+ sourceIds,
754
+ userSeq: Number.isFinite(supplied.userSeq) ? supplied.userSeq : bucketSequence(user),
755
+ index,
756
+ messages,
757
+ text,
758
+ tokens: estimateMessagesTokens(text),
759
+ };
760
+ }
761
+
762
+ function splitBucketTurns(snapshot) {
763
+ const turns = [];
764
+ let messages = [];
765
+ let keys = new Set();
766
+ let startIndex = 0;
767
+ for (let index = 0; index < snapshot.length; index += 1) {
768
+ const message = snapshot[index];
769
+ if (bucketUserBoundary(message)) {
770
+ const nextKeys = bucketUserKeys(message);
771
+ const sameTurn = nextKeys.some(key => keys.has(key));
772
+ if (messages.length && !sameTurn) {
773
+ turns.push(bucketTurn(messages, startIndex));
774
+ messages = [];
775
+ keys = new Set();
776
+ }
777
+ if (!messages.length) startIndex = index;
778
+ nextKeys.forEach(key => keys.add(key));
779
+ }
780
+ // A leading assistant fragment cannot become a complete historical turn.
781
+ if (messages.length || bucketUserBoundary(message)) messages.push(message);
782
+ }
783
+ if (messages.length) turns.push(bucketTurn(messages, startIndex));
784
+ return turns;
785
+ }
786
+
787
+ function bucketOverlap(left, right) {
788
+ const ids = new Set(left.sourceIds);
789
+ const keys = new Set(left.keys);
790
+ return right.sourceIds.some(id => ids.has(id)) || right.keys.some(key => keys.has(key));
791
+ }
792
+
793
+ function bucketCap(value, fallback, maximum = Infinity) {
794
+ return Number.isFinite(value) && value >= 0
795
+ ? Math.min(maximum, Math.floor(value)) : fallback;
796
+ }
797
+
798
+ function scoreBucketTurn(turn, options, stats = {}) {
799
+ // Relevance belongs to conversation/recall-relevance, not the budget layer.
800
+ // Indexed candidates already carry scores; raw/evicted turns use injection.
801
+ const result = typeof options.scoreTurn === 'function'
802
+ ? options.scoreTurn({
803
+ id: turn.id, userSeq: turn.userSeq, messages: turn.text,
804
+ sourceMessageIds: turn.sourceIds,
805
+ }, options.prompt || '')
806
+ : Number.isFinite(turn.score)
807
+ ? { score: turn.score, matchedTerms: turn.matchedTerms }
808
+ : scoreRecallTurn(options.prompt || '', turn.text.map(message => (
809
+ typeof message.content === 'string' ? message.content
810
+ : Array.isArray(message.content) ? message.content.map(part => (
811
+ typeof part === 'string' ? part : part?.text || ''
812
+ )).join('\n') : ''
813
+ )).join('\n'), stats);
814
+ const score = typeof result === 'number' ? result : result?.score;
815
+ return {
816
+ ...turn,
817
+ score: Number.isFinite(score) && score > 0 && result?.eligible !== false ? score : 0,
818
+ matchedTerms: Array.isArray(result?.matchedTerms) ? result.matchedTerms : [],
819
+ };
820
+ }
821
+
822
+ function bucketBefore(candidate, boundary) {
823
+ if (!boundary) return true;
824
+ if (candidate.index != null && boundary.index != null) return candidate.index < boundary.index;
825
+ return Number.isFinite(candidate.userSeq) && Number.isFinite(boundary.userSeq)
826
+ && candidate.userSeq < boundary.userSeq;
827
+ }
828
+
829
+ function describeBucket(turns, messages = turns.flatMap(turn => turn.text)) {
830
+ return {
831
+ turnCount: turns.length,
832
+ turnIds: turns.map(turn => turn.id),
833
+ sourceMessageIds: [...new Set(turns.flatMap(turn => turn.sourceIds))],
834
+ tokenCount: estimateMessagesTokens(messages),
835
+ messageCount: messages.length,
836
+ };
837
+ }
838
+
839
+ /**
840
+ * Recompute provider history from untrimmed candidates; never mutate/cache the
841
+ * result in the transcript. Past human turns are atomic text units, including
842
+ * every VP's assistant text. Tools are optional enrichment, newest first.
843
+ *
844
+ * The active turn is outside both buckets and consumes the global budget first.
845
+ * Its opening user row is protected; an oversized active turn alone may be
846
+ * fitted using the legacy protocol-safe transform. Past turns are never fitted.
847
+ * External recall must have comparable userSeq/source identities to establish
848
+ * that it predates recent/current history; unknown chronology fails closed.
849
+ *
850
+ * @param {Array<object>} snapshot Untrimmed history plus the active execution.
851
+ * @param {{ prompt?: string, relatedTurns?: Array<object>, recentTurnCap?: number,
852
+ * relatedTurnCap?: number, messageTokenBudget?: number, maxMessageCount?: number,
853
+ * language?: string, currentTurnStartIndex?: number,
854
+ * scoreTurn?: (turn: object, prompt: string) => (number|object) }} [options]
855
+ * @returns {{messages: Array<object>, meta: object}}
856
+ */
857
+ export function buildHistoryBuckets(snapshot, options = {}) {
858
+ const source = Array.isArray(snapshot) ? snapshot : [];
859
+ const tokenBudget = bucketCap(options.messageTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET);
860
+ const messageCap = bucketCap(options.maxMessageCount, DEFAULT_RUNTIME_CACHE_MESSAGE_CAP);
861
+ const recentCap = bucketCap(options.recentTurnCap, 20);
862
+ const relatedCap = bucketCap(options.relatedTurnCap, 8, 10);
863
+ const allTurns = splitBucketTurns(source);
864
+ const currentStart = Number.isInteger(options.currentTurnStartIndex)
865
+ ? Math.max(0, Math.min(source.length, options.currentTurnStartIndex))
866
+ : (allTurns.at(-1)?.index ?? source.length);
867
+ const currentSource = source.slice(currentStart);
868
+ const currentIdentity = bucketTurn(currentSource, currentStart);
869
+ let current = [];
870
+ if (currentSource.length && tokenBudget >= 2 && messageCap > 0) {
871
+ // Reserve the opening prompt before fitting later active execution units.
872
+ const first = shrinkMessageToBudget(stripAllToolNoise([currentSource[0]])[0], tokenBudget);
873
+ if (first && estimateMessageTokens(first) <= tokenBudget) current.push(first);
874
+ const remainingTokens = tokenBudget - estimateMessagesTokens(current);
875
+ const remainingRows = messageCap - current.length;
876
+ if (remainingTokens >= 2 && remainingRows > 0) {
877
+ // Active execution is not visible historical text: internal completion
878
+ // notices must reach the next provider call. Fit newest protocol units
879
+ // directly, without legacy human-turn filtering or text projection.
880
+ const units = providerUnits(pairSanitize(truncateToolResultsForModel(
881
+ currentSource.slice(1), { language: options.language },
882
+ )));
883
+ const fitted = [];
884
+ let tokens = remainingTokens;
885
+ let rows = remainingRows;
886
+ for (let index = units.length - 1; index >= 0; index -= 1) {
887
+ const unit = fitProviderUnit(units[index], tokens);
888
+ const cost = estimateMessagesTokens(unit);
889
+ if (unit.length > rows || cost > tokens) continue;
890
+ fitted.unshift(unit);
891
+ tokens -= cost;
892
+ rows -= unit.length;
893
+ }
894
+ current.push(...fitted.flat());
895
+ }
896
+ // The legacy fitter assumes a normal positive budget; at tiny allowances
897
+ // even an empty row's framing can exceed it. Remove complete tail units.
898
+ while (estimateMessagesTokens(current) > tokenBudget || current.length > messageCap) {
899
+ current = pairSanitize(current.slice(0, -1));
900
+ }
901
+ }
902
+ const availableTokens = Math.max(0, tokenBudget - estimateMessagesTokens(current));
903
+ const availableRows = Math.max(0, messageCap - current.length);
904
+ let duplicateCount = 0;
905
+ const past = [];
906
+ for (const turn of splitBucketTurns(source.slice(0, currentStart))) {
907
+ if (bucketOverlap(turn, currentIdentity)) {
908
+ duplicateCount += 1;
909
+ continue;
910
+ }
911
+ const previous = past.findIndex(other => turn.keys.some(key => other.keys.includes(key)));
912
+ if (previous >= 0) {
913
+ // Fan-out may return to the same human question after another question.
914
+ // Keep the first user boundary's position, with all complementary replies
915
+ // in their source order. Only repeated stable message identities disappear.
916
+ const original = past[previous];
917
+ const seen = new Set();
918
+ const messages = [...original.messages, ...turn.messages].filter(message => {
919
+ const ids = bucketSourceIds(message);
920
+ const repeated = ids.length > 0 && ids.every(id => seen.has(id));
921
+ ids.forEach(id => seen.add(id));
922
+ return !repeated;
923
+ });
924
+ past[previous] = bucketTurn(messages, original.index);
925
+ duplicateCount += 1;
926
+ } else past.push(turn);
927
+ }
928
+ let candidates = past.map(turn => scoreBucketTurn(turn, options));
929
+ if (typeof options.scoreTurn !== 'function') {
930
+ // Evicted in-memory turns pass the same distinctiveness gate as indexed
931
+ // turns. Otherwise ubiquitous topic words bypass the index rejection.
932
+ const termDocumentFrequency = Object.create(null);
933
+ for (const turn of candidates) {
934
+ for (const term of turn.matchedTerms) {
935
+ const key = term.toLocaleLowerCase();
936
+ termDocumentFrequency[key] = (termDocumentFrequency[key] || 0) + 1;
937
+ }
938
+ }
939
+ const stats = { sampleSize: past.length, termDocumentFrequency };
940
+ candidates = past.map(turn => scoreBucketTurn(turn, options, stats));
941
+ }
942
+ for (const entry of Array.isArray(options.relatedTurns) ? options.relatedTurns : []) {
943
+ if (!Array.isArray(entry?.messages) || !entry.messages.some(bucketUserBoundary)) continue;
944
+ const turn = bucketTurn(entry.messages, null, entry);
945
+ if (bucketOverlap(turn, currentIdentity)) { duplicateCount += 1; continue; }
946
+ const duplicate = candidates.find(other => bucketOverlap(turn, other));
947
+ if (duplicate) {
948
+ duplicateCount += 1;
949
+ // Preserve the complete untrimmed snapshot turn, but retain the index's
950
+ // qualified score when no raw-turn scorer is installed.
951
+ if (typeof options.scoreTurn !== 'function' && entry.score > (duplicate.score || 0)) {
952
+ duplicate.score = entry.score;
953
+ duplicate.matchedTerms = entry.matchedTerms || [];
954
+ }
955
+ continue;
956
+ }
957
+ candidates.push(scoreBucketTurn(turn, options));
958
+ }
959
+ const eligible = candidates.filter(turn => turn.score > 0 && turn.text.length
960
+ && (currentSource.length === 0 || bucketBefore(turn, currentIdentity)
961
+ // A runtime prompt can lack its persisted sequence. An older persisted
962
+ // past-turn boundary is still a safe fence; never guess from text/time.
963
+ || (turn.index == null && currentIdentity.userSeq == null && past.length > 0
964
+ && bucketBefore(turn, past.at(-1)))));
965
+ // A candidate inside the initial recent cap can become related after budget
966
+ // eviction. Reserve before picking the final suffix so it can re-enter.
967
+ const reservable = relatedCap > 0 ? eligible.filter(turn => turn.tokens <= availableTokens
968
+ && turn.text.length <= availableRows
969
+ && (recentCap === 0 || !past.length || bucketBefore(turn, past.at(-1)))) : [];
970
+ const reserve = reservable.length ? Math.floor(availableTokens * 0.25) : 0;
971
+ const reservedRows = reservable.length
972
+ ? Math.max(Math.floor(availableRows * 0.25), Math.min(...reservable.map(turn => turn.text.length))) : 0;
973
+ function selectRecent(limit, rowLimit = availableRows) {
974
+ const selected = [];
975
+ let tokens = 0;
976
+ let rows = 0;
977
+ for (let index = past.length - 1; index >= 0 && selected.length < recentCap; index -= 1) {
978
+ const turn = past[index];
979
+ if (tokens + turn.tokens > limit || rows + turn.text.length > rowLimit) break;
980
+ selected.unshift(turn);
981
+ tokens += turn.tokens;
982
+ rows += turn.text.length;
983
+ }
984
+ return selected;
985
+ }
986
+ let recent = selectRecent(availableTokens - reserve, availableRows - reservedRows);
987
+ let remainingTokens = availableTokens - recent.reduce((total, turn) => total + turn.tokens, 0);
988
+ let remainingRows = availableRows - recent.reduce((total, turn) => total + turn.text.length, 0);
989
+ const related = [];
990
+ const ranked = eligible.slice().sort((a, b) => b.score - a.score
991
+ || (a.userSeq ?? a.index ?? 0) - (b.userSeq ?? b.index ?? 0));
992
+ for (const turn of ranked) {
993
+ if (related.length >= relatedCap) break;
994
+ if (!bucketBefore(turn, recent[0]) || recent.some(other => bucketOverlap(turn, other))
995
+ || related.some(other => bucketOverlap(turn, other)
996
+ // Mixed-source turns require a proven order in either direction.
997
+ // Unknown sequence is not zero; skip rather than invent chronology.
998
+ || (!bucketBefore(turn, other) && !bucketBefore(other, turn)))) continue;
999
+ if (turn.tokens > remainingTokens || turn.text.length > remainingRows) continue;
1000
+ related.push(turn);
1001
+ remainingTokens -= turn.tokens;
1002
+ remainingRows -= turn.text.length;
1003
+ }
1004
+ if (!related.length) recent = selectRecent(availableTokens);
1005
+ else {
1006
+ // Pay the actual related cost, not the provisional reserve. Expand the
1007
+ // recent suffix only while every related turn remains strictly older.
1008
+ const expanded = selectRecent(
1009
+ availableTokens - related.reduce((sum, turn) => sum + turn.tokens, 0),
1010
+ availableRows - related.reduce((sum, turn) => sum + turn.text.length, 0),
1011
+ );
1012
+ while (expanded.length > recent.length && related.some(turn => (
1013
+ !bucketBefore(turn, expanded[0]) || bucketOverlap(turn, expanded[0])
1014
+ ))) expanded.shift();
1015
+ if (expanded.length > recent.length) recent = expanded;
1016
+ }
1017
+ related.sort((a, b) => bucketBefore(a, b) ? -1 : bucketBefore(b, a) ? 1 : 0);
1018
+
1019
+ const relatedMessages = related.flatMap(turn => turn.text);
1020
+ const recentText = withHistorySourceIndexes(recent.flatMap(turn => turn.messages)
1021
+ .filter(isVisibleConversationRow));
1022
+ const recentBaseline = bucketTextMessages(recentText);
1023
+ // Enrich only after both complete-text buckets and the active turn are paid.
1024
+ const recentMessages = withoutHistorySourceIndexes(addOptionalRecentToolPairs(
1025
+ recentBaseline,
1026
+ truncateToolResultsForModel(recentText, { language: options.language }),
1027
+ {
1028
+ messageTokenBudget: availableTokens - estimateMessagesTokens(relatedMessages),
1029
+ maxMessageCount: availableRows - relatedMessages.length,
1030
+ },
1031
+ ));
1032
+ const messages = [...relatedMessages.map(message => ({ ...message })), ...recentMessages, ...current];
1033
+ const retained = [...related, ...recent];
1034
+ const droppedTurns = past.filter(turn => !retained.some(other => bucketOverlap(turn, other)
1035
+ || turn === other));
1036
+ return {
1037
+ messages,
1038
+ meta: {
1039
+ recent: describeBucket(recent, recentMessages),
1040
+ related: { ...describeBucket(related), scores: related.map(turn => ({
1041
+ id: turn.id, score: turn.score, matchedTerms: turn.matchedTerms,
1042
+ })) },
1043
+ current: {
1044
+ ...describeBucket(currentSource.length ? [currentIdentity] : [], current),
1045
+ startIndex: currentStart,
1046
+ originalTokenCount: estimateMessagesTokens(currentSource),
1047
+ },
1048
+ budget: {
1049
+ messageTokenBudget: tokenBudget, maxMessageCount: messageCap,
1050
+ recentTurnCap: recentCap, relatedTurnCap: relatedCap,
1051
+ relatedReservedTokens: reserve, availableHistoryTokens: availableTokens,
1052
+ usedTokens: estimateMessagesTokens(messages), usedMessages: messages.length,
1053
+ },
1054
+ dropped: {
1055
+ pastTurnCount: droppedTurns.length,
1056
+ sourceMessageIds: [...new Set(droppedTurns.flatMap(turn => turn.sourceIds))],
1057
+ duplicateTurnCount: duplicateCount,
1058
+ oversizedTurnCount: candidates.filter(turn => turn.tokens > availableTokens
1059
+ || turn.text.length > availableRows).length,
1060
+ unselectedRelatedTurnCount: eligible.length - related.length,
1061
+ },
1062
+ },
1063
+ };
1064
+ }
1065
+
696
1066
  /**
697
1067
  * Bound the Session-level runtime history cache. This is deliberately stricter
698
1068
  * than the provider configuration: the cache is only a disposable source