@yeaft/webchat-agent 1.0.509 → 1.0.511
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +127 -122
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/cli.js +1 -1
- package/yeaft/config.js +1 -1
- package/yeaft/conversation/persist.js +2 -2
- package/yeaft/engine.js +235 -46
- package/yeaft/history-window.js +168 -55
- package/yeaft/post-compact.js +76 -0
- package/yeaft/session.js +1 -1
- package/yeaft/tool-folding/index.js +9 -15
- package/yeaft/tool-folding/t1-reflector.js +3 -2
- package/yeaft/web-bridge.js +1 -1
package/yeaft/history-window.js
CHANGED
|
@@ -27,7 +27,8 @@ export const DEFAULT_RUNTIME_CACHE_TURN_CAP = 25;
|
|
|
27
27
|
export const DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET = 32768;
|
|
28
28
|
export const DEFAULT_RUNTIME_CACHE_MESSAGE_CAP = 256;
|
|
29
29
|
|
|
30
|
-
const
|
|
30
|
+
const DEFAULT_PROVIDER_RECENT_TURNS = 10;
|
|
31
|
+
const MINIMUM_RECENT_PROVIDER_TURNS = 3;
|
|
31
32
|
const IMAGE_PART_TOKEN_COST = 1024;
|
|
32
33
|
const DOCUMENT_PART_TOKEN_COST = 2048;
|
|
33
34
|
const CONTENT_PART_FRAME_TOKENS = 2;
|
|
@@ -894,16 +895,56 @@ function describeBucket(turns, messages = turns.flatMap(turn => turn.text)) {
|
|
|
894
895
|
};
|
|
895
896
|
}
|
|
896
897
|
|
|
898
|
+
/**
|
|
899
|
+
* Emergency projection for the three-turn floor. It reuses the existing
|
|
900
|
+
* deterministic message fitter against a disposable provider copy, while
|
|
901
|
+
* sharing rows/tokens across the newest three human boundaries. The durable
|
|
902
|
+
* transcript is never rewritten.
|
|
903
|
+
*/
|
|
904
|
+
function compressRecentTurns(turns, tokenBudget, messageCap) {
|
|
905
|
+
const fitted = [];
|
|
906
|
+
let tokens = tokenBudget;
|
|
907
|
+
let rows = messageCap;
|
|
908
|
+
for (let index = 0; index < turns.length; index += 1) {
|
|
909
|
+
const turn = turns[index];
|
|
910
|
+
const remaining = turns.length - index;
|
|
911
|
+
let turnTokens = Math.floor(tokens / remaining);
|
|
912
|
+
const minimumRowTokens = Array.isArray(turn.text[0]?.content) ? 5 : 3;
|
|
913
|
+
const turnRows = Math.min(Math.floor(rows / remaining), Math.floor(turnTokens / minimumRowTokens));
|
|
914
|
+
if (turnTokens < minimumRowTokens || turnRows < 1) continue;
|
|
915
|
+
const selected = turn.text.length <= turnRows ? turn.text
|
|
916
|
+
: [turn.text[0], ...(turnRows > 1 ? turn.text.slice(-(turnRows - 1)) : [])];
|
|
917
|
+
const text = [];
|
|
918
|
+
for (let row = 0; row < selected.length; row += 1) {
|
|
919
|
+
const allowance = Math.floor(turnTokens / (selected.length - row));
|
|
920
|
+
const message = shrinkMessageToBudget(selected[row], allowance);
|
|
921
|
+
const cost = estimateMessageTokens(message);
|
|
922
|
+
if (cost <= allowance && hasProviderContent(message.content)) {
|
|
923
|
+
text.push(message);
|
|
924
|
+
turnTokens -= cost;
|
|
925
|
+
} else if (row === 0) break;
|
|
926
|
+
}
|
|
927
|
+
if (!text.some(bucketUserBoundary)) continue;
|
|
928
|
+
const cost = estimateMessagesTokens(text);
|
|
929
|
+
fitted.push({ ...turn, text, tokens: cost });
|
|
930
|
+
tokens -= cost;
|
|
931
|
+
rows -= text.length;
|
|
932
|
+
}
|
|
933
|
+
return fitted;
|
|
934
|
+
}
|
|
935
|
+
|
|
897
936
|
/**
|
|
898
937
|
* Recompute provider history from untrimmed candidates; never mutate/cache the
|
|
899
938
|
* result in the transcript. Past human turn boundaries are retained when the
|
|
900
939
|
* configured recent window fits; tools are optional enrichment, newest first.
|
|
901
940
|
*
|
|
902
|
-
* The active turn is outside both buckets and
|
|
903
|
-
*
|
|
904
|
-
*
|
|
905
|
-
*
|
|
906
|
-
*
|
|
941
|
+
* The active turn is outside both buckets and outside the history budget. Its
|
|
942
|
+
* opening user row is protected by the later whole-request fitter. Prefer ten
|
|
943
|
+
* complete recent turns, reducing the oldest end down to three. If that floor
|
|
944
|
+
* cannot fit, omit recall and compress only those three turns; historical tool
|
|
945
|
+
* replay then narrows from the normal newest three turns to the newest one.
|
|
946
|
+
* Related recall otherwise uses remaining history budget and stays optional and
|
|
947
|
+
* complete. External recall must have
|
|
907
948
|
* comparable userSeq/source identities to establish
|
|
908
949
|
* that it predates recent/current history; unknown chronology fails closed.
|
|
909
950
|
*
|
|
@@ -918,7 +959,7 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
918
959
|
const source = Array.isArray(snapshot) ? snapshot : [];
|
|
919
960
|
const tokenBudget = bucketCap(options.messageTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET);
|
|
920
961
|
const messageCap = bucketCap(options.maxMessageCount, DEFAULT_RUNTIME_CACHE_MESSAGE_CAP);
|
|
921
|
-
const recentCap = bucketCap(options.recentTurnCap,
|
|
962
|
+
const recentCap = bucketCap(options.recentTurnCap, DEFAULT_PROVIDER_RECENT_TURNS);
|
|
922
963
|
const relatedCap = bucketCap(options.relatedTurnCap, 5, 5);
|
|
923
964
|
const keepToolTurns = bucketCap(options.keepToolTurns, DEFAULT_KEEP_TOOL_TURNS);
|
|
924
965
|
const allTurns = splitBucketTurns(source);
|
|
@@ -927,41 +968,16 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
927
968
|
: (allTurns.at(-1)?.index ?? source.length);
|
|
928
969
|
const currentSource = source.slice(currentStart);
|
|
929
970
|
const currentIdentity = bucketTurn(currentSource, currentStart);
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
// directly, without legacy human-turn filtering or text projection.
|
|
941
|
-
const units = providerUnits(pairSanitize(truncateToolResultsForModel(
|
|
942
|
-
currentSource.slice(1), { language: options.language },
|
|
943
|
-
)));
|
|
944
|
-
const fitted = [];
|
|
945
|
-
let tokens = remainingTokens;
|
|
946
|
-
let rows = remainingRows;
|
|
947
|
-
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
948
|
-
const unit = fitProviderUnit(units[index], tokens);
|
|
949
|
-
const cost = estimateMessagesTokens(unit);
|
|
950
|
-
if (unit.length > rows || cost > tokens) continue;
|
|
951
|
-
fitted.unshift(unit);
|
|
952
|
-
tokens -= cost;
|
|
953
|
-
rows -= unit.length;
|
|
954
|
-
}
|
|
955
|
-
current.push(...fitted.flat());
|
|
956
|
-
}
|
|
957
|
-
// The legacy fitter assumes a normal positive budget; at tiny allowances
|
|
958
|
-
// even an empty row's framing can exceed it. Remove complete tail units.
|
|
959
|
-
while (estimateMessagesTokens(current) > tokenBudget || current.length > messageCap) {
|
|
960
|
-
current = pairSanitize(current.slice(0, -1));
|
|
961
|
-
}
|
|
962
|
-
}
|
|
963
|
-
const availableTokens = Math.max(0, tokenBudget - estimateMessagesTokens(current));
|
|
964
|
-
const availableRows = Math.max(0, messageCap - current.length);
|
|
971
|
+
// The 32K/default budget owns only rows before currentStart. Keep the active
|
|
972
|
+
// turn intact here; whole-request fitting runs at every provider boundary and
|
|
973
|
+
// uses the actual model context window. Tool bodies may still receive their
|
|
974
|
+
// normal deterministic per-result truncation, without touching the durable
|
|
975
|
+
// transcript or charging that copy against history.
|
|
976
|
+
const current = pairSanitize(truncateToolResultsForModel(
|
|
977
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
978
|
+
));
|
|
979
|
+
const availableTokens = tokenBudget;
|
|
980
|
+
const availableRows = messageCap;
|
|
965
981
|
let duplicateCount = 0;
|
|
966
982
|
const past = [];
|
|
967
983
|
for (const turn of splitBucketTurns(source.slice(0, currentStart))) {
|
|
@@ -1023,11 +1039,12 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1023
1039
|
// past-turn boundary is still a safe fence; never guess from text/time.
|
|
1024
1040
|
|| (turn.index == null && currentIdentity.userSeq == null && past.length > 0
|
|
1025
1041
|
&& bucketBefore(turn, past.at(-1)))));
|
|
1026
|
-
// Recent text has first claim on history budget. Drop whole oldest turns
|
|
1027
|
-
//
|
|
1028
|
-
|
|
1042
|
+
// Recent text has first claim on history budget. Drop whole oldest turns down
|
|
1043
|
+
// to the three-turn floor before projecting that floor more aggressively.
|
|
1044
|
+
let recent = [];
|
|
1029
1045
|
let recentTokens = 0;
|
|
1030
1046
|
let recentRows = 0;
|
|
1047
|
+
let compressedRecentFloor = false;
|
|
1031
1048
|
const recentCandidates = recentCap > 0 ? past.slice(-recentCap) : [];
|
|
1032
1049
|
for (let index = recentCandidates.length - 1; index >= 0; index -= 1) {
|
|
1033
1050
|
const turn = recentCandidates[index];
|
|
@@ -1037,16 +1054,15 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1037
1054
|
recentTokens += turn.tokens;
|
|
1038
1055
|
recentRows += turn.text.length;
|
|
1039
1056
|
}
|
|
1040
|
-
const
|
|
1041
|
-
if (recent.length <
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
throw error;
|
|
1057
|
+
const recentFloor = Math.min(MINIMUM_RECENT_PROVIDER_TURNS, recentCandidates.length);
|
|
1058
|
+
if (recent.length < recentFloor) {
|
|
1059
|
+
compressedRecentFloor = true;
|
|
1060
|
+
recent = compressRecentTurns(recentCandidates.slice(-recentFloor), availableTokens, availableRows);
|
|
1045
1061
|
}
|
|
1046
1062
|
let remainingTokens = availableTokens - recent.reduce((total, turn) => total + turn.tokens, 0);
|
|
1047
1063
|
let remainingRows = availableRows - recent.reduce((total, turn) => total + turn.text.length, 0);
|
|
1048
1064
|
const related = [];
|
|
1049
|
-
const ranked = eligible.slice().sort((a, b) => b.score - a.score
|
|
1065
|
+
const ranked = compressedRecentFloor ? [] : eligible.slice().sort((a, b) => b.score - a.score
|
|
1050
1066
|
|| (a.userSeq ?? a.index ?? 0) - (b.userSeq ?? b.index ?? 0));
|
|
1051
1067
|
for (const turn of ranked) {
|
|
1052
1068
|
if (related.length >= relatedCap) break;
|
|
@@ -1067,8 +1083,9 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1067
1083
|
// Enrich only after both complete-text buckets and the active turn are paid.
|
|
1068
1084
|
// Tool protocol is useful only for immediate continuity; unlike visible text,
|
|
1069
1085
|
// it never reaches farther back than the configured recent tool window.
|
|
1070
|
-
const
|
|
1071
|
-
|
|
1086
|
+
const effectiveKeepToolTurns = compressedRecentFloor ? Math.min(1, keepToolTurns) : keepToolTurns;
|
|
1087
|
+
const recentToolSource = effectiveKeepToolTurns > 0
|
|
1088
|
+
? recent.slice(-effectiveKeepToolTurns).flatMap(turn => turn.messages).filter(isVisibleConversationRow)
|
|
1072
1089
|
: [];
|
|
1073
1090
|
const recentMessages = withoutHistorySourceIndexes(addOptionalRecentToolPairs(
|
|
1074
1091
|
recentBaseline,
|
|
@@ -1097,9 +1114,14 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1097
1114
|
budget: {
|
|
1098
1115
|
messageTokenBudget: tokenBudget, maxMessageCount: messageCap,
|
|
1099
1116
|
recentTurnCap: recentCap, relatedTurnCap: relatedCap,
|
|
1100
|
-
minimumRecentTurns:
|
|
1117
|
+
minimumRecentTurns: recentFloor,
|
|
1118
|
+
compressedRecentFloor,
|
|
1119
|
+
effectiveKeepToolTurns,
|
|
1101
1120
|
relatedReservedTokens: 0, availableHistoryTokens: availableTokens,
|
|
1102
|
-
usedTokens: estimateMessagesTokens(
|
|
1121
|
+
usedTokens: estimateMessagesTokens([...relatedMessages, ...recentMessages]),
|
|
1122
|
+
usedMessages: relatedMessages.length + recentMessages.length,
|
|
1123
|
+
requestTokensBeforeWholeRequestFit: estimateMessagesTokens(messages),
|
|
1124
|
+
requestMessagesBeforeWholeRequestFit: messages.length,
|
|
1103
1125
|
},
|
|
1104
1126
|
dropped: {
|
|
1105
1127
|
pastTurnCount: droppedTurns.length,
|
|
@@ -1113,6 +1135,97 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1113
1135
|
};
|
|
1114
1136
|
}
|
|
1115
1137
|
|
|
1138
|
+
/**
|
|
1139
|
+
* Fit one provider-request copy to the actual model window. The caller tells
|
|
1140
|
+
* us where current-turn rows begin; only the prefix is subject to the history
|
|
1141
|
+
* budget. If the complete request is still too large, old history disappears
|
|
1142
|
+
* first, followed by the oldest disposable current-turn protocol units. The
|
|
1143
|
+
* source array and durable transcript are never mutated.
|
|
1144
|
+
*
|
|
1145
|
+
* @param {Array<object>} messages
|
|
1146
|
+
* @param {{ contextWindow:number, systemTokens?:number, toolSchemaTokens?:number,
|
|
1147
|
+
* outputReserve?:number, historyMessageCount?:number, historyTokenBudget?:number,
|
|
1148
|
+
* maxMessageCount?:number, language?:string }} options
|
|
1149
|
+
* @returns {{messages:Array<object>, meta:object}}
|
|
1150
|
+
*/
|
|
1151
|
+
export function fitProviderRequestToContext(messages, options = {}) {
|
|
1152
|
+
const source = Array.isArray(messages) ? messages : [];
|
|
1153
|
+
const contextWindow = bucketCap(options.contextWindow, 0);
|
|
1154
|
+
const staticTokens = bucketCap(options.systemTokens, 0)
|
|
1155
|
+
+ bucketCap(options.toolSchemaTokens, 0)
|
|
1156
|
+
+ bucketCap(options.outputReserve, 0);
|
|
1157
|
+
const messageBudget = Math.max(0, contextWindow - staticTokens);
|
|
1158
|
+
const split = Math.max(0, Math.min(source.length,
|
|
1159
|
+
Number.isInteger(options.historyMessageCount) ? options.historyMessageCount : 0));
|
|
1160
|
+
// The runtime cache's 256-row cap is a history-storage concern, not a model
|
|
1161
|
+
// request limit. Current-turn tool loops may legitimately exceed it while
|
|
1162
|
+
// remaining inside the model window. Only enforce a cap when the caller
|
|
1163
|
+
// explicitly supplies one.
|
|
1164
|
+
const messageCap = options.maxMessageCount === undefined
|
|
1165
|
+
? Number.MAX_SAFE_INTEGER
|
|
1166
|
+
: bucketCap(options.maxMessageCount, Number.MAX_SAFE_INTEGER);
|
|
1167
|
+
const historySource = source.slice(0, split);
|
|
1168
|
+
const currentSource = source.slice(split);
|
|
1169
|
+
|
|
1170
|
+
let current = pairSanitize(truncateToolResultsForModel(
|
|
1171
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
1172
|
+
));
|
|
1173
|
+
if (estimateMessagesTokens(current) > messageBudget || current.length > messageCap) {
|
|
1174
|
+
const fitted = [];
|
|
1175
|
+
if (current.length > 0 && messageBudget >= 2 && messageCap > 0) {
|
|
1176
|
+
const first = shrinkMessageToBudget(stripAllToolNoise([current[0]])[0], messageBudget);
|
|
1177
|
+
if (first && estimateMessageTokens(first) <= messageBudget) fitted.push(first);
|
|
1178
|
+
let tokens = messageBudget - estimateMessagesTokens(fitted);
|
|
1179
|
+
let rows = messageCap - fitted.length;
|
|
1180
|
+
const units = providerUnits(pairSanitize(current.slice(1)));
|
|
1181
|
+
const tail = [];
|
|
1182
|
+
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
1183
|
+
const unit = fitProviderUnit(units[index], tokens);
|
|
1184
|
+
const cost = estimateMessagesTokens(unit);
|
|
1185
|
+
if (unit.length > rows || cost > tokens) continue;
|
|
1186
|
+
tail.unshift(unit);
|
|
1187
|
+
tokens -= cost;
|
|
1188
|
+
rows -= unit.length;
|
|
1189
|
+
}
|
|
1190
|
+
fitted.push(...tail.flat());
|
|
1191
|
+
}
|
|
1192
|
+
current = pairSanitize(fitted);
|
|
1193
|
+
}
|
|
1194
|
+
|
|
1195
|
+
const configuredHistoryBudget = bucketCap(
|
|
1196
|
+
options.historyTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET,
|
|
1197
|
+
);
|
|
1198
|
+
const remainingTokens = Math.max(0, Math.min(
|
|
1199
|
+
configuredHistoryBudget,
|
|
1200
|
+
messageBudget - estimateMessagesTokens(current),
|
|
1201
|
+
));
|
|
1202
|
+
const remainingRows = Math.max(0, messageCap - current.length);
|
|
1203
|
+
const history = remainingTokens >= 2 && remainingRows > 0
|
|
1204
|
+
? trimSnapshotForBudget(historySource, {
|
|
1205
|
+
messageTokenBudget: remainingTokens,
|
|
1206
|
+
maxMessageCount: remainingRows,
|
|
1207
|
+
recentTurnCap: Number.MAX_SAFE_INTEGER,
|
|
1208
|
+
language: options.language,
|
|
1209
|
+
})
|
|
1210
|
+
: [];
|
|
1211
|
+
const fittedMessages = [...history, ...current];
|
|
1212
|
+
return {
|
|
1213
|
+
messages: fittedMessages,
|
|
1214
|
+
meta: {
|
|
1215
|
+
contextWindow,
|
|
1216
|
+
staticTokens,
|
|
1217
|
+
messageBudget,
|
|
1218
|
+
estimatedTokens: staticTokens + estimateMessagesTokens(fittedMessages),
|
|
1219
|
+
historyMessagesBefore: historySource.length,
|
|
1220
|
+
historyMessagesAfter: history.length,
|
|
1221
|
+
currentMessagesBefore: currentSource.length,
|
|
1222
|
+
currentMessagesAfter: current.length,
|
|
1223
|
+
droppedHistoryMessages: historySource.length - history.length,
|
|
1224
|
+
droppedCurrentMessages: currentSource.length - current.length,
|
|
1225
|
+
},
|
|
1226
|
+
};
|
|
1227
|
+
}
|
|
1228
|
+
|
|
1116
1229
|
/**
|
|
1117
1230
|
* Bound the Session-level runtime history cache. This is deliberately stricter
|
|
1118
1231
|
* than the provider configuration: the cache is only a disposable source
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Non-blocking, post-response conversation compaction.
|
|
3
|
+
*
|
|
4
|
+
* The compact artifact is a derived provider-context cache. It never replaces
|
|
5
|
+
* or tombstones ConversationStore rows. A generation fence in Engine decides
|
|
6
|
+
* whether a completed artifact is still current before this module writes it.
|
|
7
|
+
*/
|
|
8
|
+
import { promises as fs } from 'fs';
|
|
9
|
+
import { dirname, join } from 'path';
|
|
10
|
+
|
|
11
|
+
export const POST_COMPACT_CONTEXT_RATIO = 0.8;
|
|
12
|
+
|
|
13
|
+
function safePart(value, fallback) {
|
|
14
|
+
const text = typeof value === 'string' && value.trim() ? value.trim() : fallback;
|
|
15
|
+
return encodeURIComponent(text).replace(/%/g, '_');
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function postCompactPath(yeaftDir, { sessionId, vpId, threadId } = {}) {
|
|
19
|
+
if (!yeaftDir || !sessionId) return null;
|
|
20
|
+
const file = `${safePart(vpId, 'default')}--${safePart(threadId, 'main')}.json`;
|
|
21
|
+
return join(yeaftDir, 'sessions', safePart(sessionId, 'session'), 'conversation', 'post-compact', file);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export async function loadPostCompact(path) {
|
|
25
|
+
if (!path) return null;
|
|
26
|
+
try {
|
|
27
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
28
|
+
return parsed && parsed.version === 1 && typeof parsed.summary === 'string'
|
|
29
|
+
? parsed : null;
|
|
30
|
+
} catch {
|
|
31
|
+
return null;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export async function savePostCompact(path, artifact, isCurrent = null) {
|
|
36
|
+
if (!path) return false;
|
|
37
|
+
await fs.mkdir(dirname(path), { recursive: true });
|
|
38
|
+
const temp = `${path}.${process.pid}.${Date.now()}.tmp`;
|
|
39
|
+
await fs.writeFile(temp, `${JSON.stringify({ version: 1, ...artifact }, null, 2)}\n`, 'utf8');
|
|
40
|
+
if (typeof isCurrent === 'function' && !isCurrent()) {
|
|
41
|
+
await fs.unlink(temp).catch(() => {});
|
|
42
|
+
return false;
|
|
43
|
+
}
|
|
44
|
+
await fs.rename(temp, path);
|
|
45
|
+
return true;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export async function removePostCompactIfSource(path, sourceTurnId) {
|
|
49
|
+
if (!path || !sourceTurnId) return false;
|
|
50
|
+
try {
|
|
51
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
52
|
+
if (parsed?.sourceTurnId !== sourceTurnId) return false;
|
|
53
|
+
await fs.unlink(path);
|
|
54
|
+
return true;
|
|
55
|
+
} catch {
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export async function generatePostCompact({ adapter, model, messages, maxTokens = 4096 }) {
|
|
61
|
+
const transcript = JSON.stringify((Array.isArray(messages) ? messages : []).map(message => ({
|
|
62
|
+
role: message?.role,
|
|
63
|
+
content: message?.content,
|
|
64
|
+
...(Array.isArray(message?.toolCalls) ? { toolCalls: message.toolCalls } : {}),
|
|
65
|
+
...(message?.toolCallId ? { toolCallId: message.toolCallId } : {}),
|
|
66
|
+
})));
|
|
67
|
+
const result = await adapter.call({
|
|
68
|
+
model,
|
|
69
|
+
system: 'Summarize the earlier conversation for use as context in a later turn. Preserve user goals, decisions, constraints, unresolved work, and important results. Omit raw tool payloads and do not invent facts. Return only the compact summary.',
|
|
70
|
+
messages: [{ role: 'user', content: `Compact this transcript:\n${transcript}` }],
|
|
71
|
+
maxTokens,
|
|
72
|
+
});
|
|
73
|
+
const summary = typeof result?.text === 'string' ? result.text.trim() : '';
|
|
74
|
+
if (!summary) throw new Error('post compact returned empty content');
|
|
75
|
+
return summary;
|
|
76
|
+
}
|
package/yeaft/session.js
CHANGED
|
@@ -180,7 +180,7 @@ export async function loadSession(options = {}) {
|
|
|
180
180
|
if (typeof dreamEnabled === 'boolean') config.dream.enabled = dreamEnabled;
|
|
181
181
|
|
|
182
182
|
// Propagate the (clamped) cold-start replay window to the conversation
|
|
183
|
-
// store. The default is
|
|
183
|
+
// store. The default is 10 turns; a user wanting more recall after a
|
|
184
184
|
// fresh boot sets `yeaft.recentTurnsLimit` in ~/.yeaft/config.json.
|
|
185
185
|
// Called once per session boot — subsequent boots overwrite the
|
|
186
186
|
// module-level default safely (single-process model).
|
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
* tool-folding/index.js — V7 reflection subsystem entry (PR-L).
|
|
3
3
|
*
|
|
4
4
|
* Exposes:
|
|
5
|
-
* - Constants
|
|
5
|
+
* - Constants TOOL_LOOP_REFLECTION_INTERVAL, TURN_SUMMARY_THRESHOLD,
|
|
6
|
+
* DUP_TOOL_THRESHOLD
|
|
6
7
|
* - Reflector helpers (T1 sync, T2 async, fallback stub)
|
|
7
8
|
* - Helpers for collapsing message ranges into a single assistant
|
|
8
9
|
* reflection message
|
|
@@ -10,19 +11,10 @@
|
|
|
10
11
|
*
|
|
11
12
|
* The constants are NOT config-driven — V7 design freezes them in code.
|
|
12
13
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* end-of-turn reflection path. Keep a usefully wide gap between the two
|
|
18
|
-
* so the (T2, T1) band where T2-alone applies stays meaningful.
|
|
19
|
-
*
|
|
20
|
-
* TOOL_BATCH_SIZE history: was 13 originally; raised to 30 (2026-05-15)
|
|
21
|
-
* after user feedback that 13 fired too often inside a single task and
|
|
22
|
-
* fragmented otherwise-coherent tool arcs into multiple reflections. 30
|
|
23
|
-
* keeps the periodic-reflection contract (it still fires every N tools,
|
|
24
|
-
* not just once) but gives a single task arc room to breathe before the
|
|
25
|
-
* arc gets collapsed.
|
|
14
|
+
* T1 runs inside the turn and collapses history in place. Its cadence is
|
|
15
|
+
* measured in provider tool loops (assistant tool_use batch → execution →
|
|
16
|
+
* next provider boundary), not in the number of calls inside a batch. A model
|
|
17
|
+
* returning 30 parallel tools has completed one loop, not thirty.
|
|
26
18
|
*
|
|
27
19
|
* TURN_SUMMARY_THRESHOLD history: was 5 originally; raised to 8
|
|
28
20
|
* (2026-05-18). 5 was too aggressive — small "read a few files, edit one,
|
|
@@ -32,7 +24,9 @@
|
|
|
32
24
|
* before the next turn's history grows.
|
|
33
25
|
*/
|
|
34
26
|
|
|
35
|
-
export const
|
|
27
|
+
export const TOOL_LOOP_REFLECTION_INTERVAL = 30;
|
|
28
|
+
// Compatibility for external imports; the engine uses the loop-specific name.
|
|
29
|
+
export const TOOL_BATCH_SIZE = TOOL_LOOP_REFLECTION_INTERVAL;
|
|
36
30
|
export const TURN_SUMMARY_THRESHOLD = 8;
|
|
37
31
|
export const DUP_TOOL_THRESHOLD = 3;
|
|
38
32
|
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* t1-reflector.js — V7 in-turn (synchronous) reflection (PR-L).
|
|
3
3
|
*
|
|
4
|
-
* Triggered
|
|
5
|
-
*
|
|
4
|
+
* Triggered after each interval of 30 completed tool loops, immediately before
|
|
5
|
+
* the engine loops back into adapter.stream(). Parallel calls returned in one
|
|
6
|
+
* assistant tool-use batch count as one loop. Calls
|
|
6
7
|
* the PRIMARY model — never the fast model — to generate a markdown
|
|
7
8
|
* reflection over the batch.
|
|
8
9
|
*
|
package/yeaft/web-bridge.js
CHANGED
|
@@ -7174,7 +7174,7 @@ export async function handleYeaftLoadHistory(msg) {
|
|
|
7174
7174
|
// `lim` is now expressed in TURNS, not raw messages. `loadRecent` and
|
|
7175
7175
|
// `loadRecentBySession` use turn-based slicing so the cut never lands
|
|
7176
7176
|
// mid-tool-arc. Pass `undefined` to use the persistence-layer default
|
|
7177
|
-
// (DEFAULT_RECENT_TURNS =
|
|
7177
|
+
// (DEFAULT_RECENT_TURNS = 10 turns).
|
|
7178
7178
|
const pickRecent = (store, lim) =>
|
|
7179
7179
|
sessionId ? store.loadRecentBySession(sessionId, lim) : store.loadRecent(lim);
|
|
7180
7180
|
let historyAlreadyReplayed = false;
|