@yeaft/webchat-agent 1.0.508 → 1.0.510
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/connection/message-router.js +1 -1
- package/local-runtime/server/handlers/agent-sync.js +1 -0
- package/local-runtime/server/handlers/client-conversation.js +38 -2
- package/local-runtime/server/handlers/client-misc.js +1 -1
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +212 -99
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/config-api.js +59 -6
- package/yeaft/config.js +3 -3
- package/yeaft/conversation/history-index-worker.js +3 -2
- package/yeaft/conversation/history-index.js +6 -4
- package/yeaft/conversation/persist.js +37 -0
- package/yeaft/conversation/recall-relevance.js +44 -11
- package/yeaft/engine.js +262 -56
- package/yeaft/history-window.js +128 -122
- package/yeaft/post-compact.js +76 -0
- package/yeaft/tool-folding/index.js +9 -15
- package/yeaft/tool-folding/t1-reflector.js +3 -2
- package/yeaft/web-bridge.js +130 -39
package/yeaft/history-window.js
CHANGED
|
@@ -27,7 +27,6 @@ export const DEFAULT_RUNTIME_CACHE_TURN_CAP = 25;
|
|
|
27
27
|
export const DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET = 32768;
|
|
28
28
|
export const DEFAULT_RUNTIME_CACHE_MESSAGE_CAP = 256;
|
|
29
29
|
|
|
30
|
-
const MINIMUM_RECENT_PROVIDER_TURNS = 20;
|
|
31
30
|
const IMAGE_PART_TOKEN_COST = 1024;
|
|
32
31
|
const DOCUMENT_PART_TOKEN_COST = 2048;
|
|
33
32
|
const CONTENT_PART_FRAME_TOKENS = 2;
|
|
@@ -899,10 +898,11 @@ function describeBucket(turns, messages = turns.flatMap(turn => turn.text)) {
|
|
|
899
898
|
* result in the transcript. Past human turn boundaries are retained when the
|
|
900
899
|
* configured recent window fits; tools are optional enrichment, newest first.
|
|
901
900
|
*
|
|
902
|
-
* The active turn is outside both buckets and
|
|
903
|
-
*
|
|
904
|
-
*
|
|
905
|
-
*
|
|
901
|
+
* The active turn is outside both buckets and outside the history budget. Its
|
|
902
|
+
* opening user row is protected by the later whole-request fitter. Recent text
|
|
903
|
+
* stays complete when it fits, but a configured turn count is never a hard
|
|
904
|
+
* request-success floor. Related recall only uses remaining history budget and
|
|
905
|
+
* stays optional and complete. External recall must have
|
|
906
906
|
* comparable userSeq/source identities to establish
|
|
907
907
|
* that it predates recent/current history; unknown chronology fails closed.
|
|
908
908
|
*
|
|
@@ -918,7 +918,7 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
918
918
|
const tokenBudget = bucketCap(options.messageTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET);
|
|
919
919
|
const messageCap = bucketCap(options.maxMessageCount, DEFAULT_RUNTIME_CACHE_MESSAGE_CAP);
|
|
920
920
|
const recentCap = bucketCap(options.recentTurnCap, 20);
|
|
921
|
-
const relatedCap = bucketCap(options.relatedTurnCap,
|
|
921
|
+
const relatedCap = bucketCap(options.relatedTurnCap, 5, 5);
|
|
922
922
|
const keepToolTurns = bucketCap(options.keepToolTurns, DEFAULT_KEEP_TOOL_TURNS);
|
|
923
923
|
const allTurns = splitBucketTurns(source);
|
|
924
924
|
const currentStart = Number.isInteger(options.currentTurnStartIndex)
|
|
@@ -926,41 +926,16 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
926
926
|
: (allTurns.at(-1)?.index ?? source.length);
|
|
927
927
|
const currentSource = source.slice(currentStart);
|
|
928
928
|
const currentIdentity = bucketTurn(currentSource, currentStart);
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
// directly, without legacy human-turn filtering or text projection.
|
|
940
|
-
const units = providerUnits(pairSanitize(truncateToolResultsForModel(
|
|
941
|
-
currentSource.slice(1), { language: options.language },
|
|
942
|
-
)));
|
|
943
|
-
const fitted = [];
|
|
944
|
-
let tokens = remainingTokens;
|
|
945
|
-
let rows = remainingRows;
|
|
946
|
-
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
947
|
-
const unit = fitProviderUnit(units[index], tokens);
|
|
948
|
-
const cost = estimateMessagesTokens(unit);
|
|
949
|
-
if (unit.length > rows || cost > tokens) continue;
|
|
950
|
-
fitted.unshift(unit);
|
|
951
|
-
tokens -= cost;
|
|
952
|
-
rows -= unit.length;
|
|
953
|
-
}
|
|
954
|
-
current.push(...fitted.flat());
|
|
955
|
-
}
|
|
956
|
-
// The legacy fitter assumes a normal positive budget; at tiny allowances
|
|
957
|
-
// even an empty row's framing can exceed it. Remove complete tail units.
|
|
958
|
-
while (estimateMessagesTokens(current) > tokenBudget || current.length > messageCap) {
|
|
959
|
-
current = pairSanitize(current.slice(0, -1));
|
|
960
|
-
}
|
|
961
|
-
}
|
|
962
|
-
const availableTokens = Math.max(0, tokenBudget - estimateMessagesTokens(current));
|
|
963
|
-
const availableRows = Math.max(0, messageCap - current.length);
|
|
929
|
+
// The 32K/default budget owns only rows before currentStart. Keep the active
|
|
930
|
+
// turn intact here; whole-request fitting runs at every provider boundary and
|
|
931
|
+
// uses the actual model context window. Tool bodies may still receive their
|
|
932
|
+
// normal deterministic per-result truncation, without touching the durable
|
|
933
|
+
// transcript or charging that copy against history.
|
|
934
|
+
const current = pairSanitize(truncateToolResultsForModel(
|
|
935
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
936
|
+
));
|
|
937
|
+
const availableTokens = tokenBudget;
|
|
938
|
+
const availableRows = messageCap;
|
|
964
939
|
let duplicateCount = 0;
|
|
965
940
|
const past = [];
|
|
966
941
|
for (const turn of splitBucketTurns(source.slice(0, currentStart))) {
|
|
@@ -1022,65 +997,20 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1022
997
|
// past-turn boundary is still a safe fence; never guess from text/time.
|
|
1023
998
|
|| (turn.index == null && currentIdentity.userSeq == null && past.length > 0
|
|
1024
999
|
&& bucketBefore(turn, past.at(-1)))));
|
|
1025
|
-
//
|
|
1026
|
-
//
|
|
1027
|
-
const
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
const
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
const remainingRows = turn.text.length - index;
|
|
1039
|
-
const allowance = Math.max(2, Math.floor(tokens / remainingRows));
|
|
1040
|
-
const message = shrinkMessageToBudget(turn.text[index], allowance);
|
|
1041
|
-
const rows = dropEmptyAssistantRows([message]);
|
|
1042
|
-
fitted.push(...rows);
|
|
1043
|
-
tokens -= estimateMessagesTokens(rows);
|
|
1044
|
-
}
|
|
1045
|
-
return fitted;
|
|
1000
|
+
// Recent text has first claim on history budget. Drop whole oldest turns,
|
|
1001
|
+
// never reserve space for recall or truncate text to manufacture turn counts.
|
|
1002
|
+
const recent = [];
|
|
1003
|
+
let recentTokens = 0;
|
|
1004
|
+
let recentRows = 0;
|
|
1005
|
+
const recentCandidates = recentCap > 0 ? past.slice(-recentCap) : [];
|
|
1006
|
+
for (let index = recentCandidates.length - 1; index >= 0; index -= 1) {
|
|
1007
|
+
const turn = recentCandidates[index];
|
|
1008
|
+
if (recentTokens + turn.tokens > availableTokens
|
|
1009
|
+
|| recentRows + turn.text.length > availableRows) break;
|
|
1010
|
+
recent.unshift(turn);
|
|
1011
|
+
recentTokens += turn.tokens;
|
|
1012
|
+
recentRows += turn.text.length;
|
|
1046
1013
|
}
|
|
1047
|
-
function selectRecent(limit, rowLimit = availableRows) {
|
|
1048
|
-
const candidates = past.slice(-recentCap);
|
|
1049
|
-
if (!candidates.length || limit <= 0 || rowLimit <= 0) return [];
|
|
1050
|
-
const preserveBoundaries = candidates.length >= MINIMUM_RECENT_PROVIDER_TURNS
|
|
1051
|
-
&& rowLimit >= candidates.reduce((sum, turn) => sum + turn.text.length, 0)
|
|
1052
|
-
&& limit >= candidates.length * 2;
|
|
1053
|
-
if (!preserveBoundaries) {
|
|
1054
|
-
const selected = [];
|
|
1055
|
-
let tokens = 0;
|
|
1056
|
-
let rows = 0;
|
|
1057
|
-
for (let index = candidates.length - 1; index >= 0; index -= 1) {
|
|
1058
|
-
const turn = candidates[index];
|
|
1059
|
-
if (tokens + turn.tokens > limit || rows + turn.text.length > rowLimit) break;
|
|
1060
|
-
selected.unshift(turn);
|
|
1061
|
-
tokens += turn.tokens;
|
|
1062
|
-
rows += turn.text.length;
|
|
1063
|
-
}
|
|
1064
|
-
return selected;
|
|
1065
|
-
}
|
|
1066
|
-
|
|
1067
|
-
const selected = [];
|
|
1068
|
-
let tokens = Math.floor(limit);
|
|
1069
|
-
let rows = Math.floor(rowLimit);
|
|
1070
|
-
for (const turn of candidates) {
|
|
1071
|
-
const remainingTurns = candidates.length - selected.length;
|
|
1072
|
-
if (rows < remainingTurns) break;
|
|
1073
|
-
const allowance = Math.max(2, Math.floor(tokens / remainingTurns));
|
|
1074
|
-
const fitted = fitRecentTurn(turn, allowance);
|
|
1075
|
-
if (!fitted.length || fitted.length > rows - (remainingTurns - 1)) continue;
|
|
1076
|
-
const used = estimateMessagesTokens(fitted);
|
|
1077
|
-
selected.push({ ...turn, text: fitted, tokens: used });
|
|
1078
|
-
tokens -= used;
|
|
1079
|
-
rows -= fitted.length;
|
|
1080
|
-
}
|
|
1081
|
-
return selected;
|
|
1082
|
-
}
|
|
1083
|
-
let recent = selectRecent(availableTokens - reserve, availableRows - reservedRows);
|
|
1084
1014
|
let remainingTokens = availableTokens - recent.reduce((total, turn) => total + turn.tokens, 0);
|
|
1085
1015
|
let remainingRows = availableRows - recent.reduce((total, turn) => total + turn.text.length, 0);
|
|
1086
1016
|
const related = [];
|
|
@@ -1098,34 +1028,15 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1098
1028
|
remainingTokens -= turn.tokens;
|
|
1099
1029
|
remainingRows -= turn.text.length;
|
|
1100
1030
|
}
|
|
1101
|
-
if (!related.length) recent = selectRecent(availableTokens);
|
|
1102
|
-
else {
|
|
1103
|
-
// Pay the actual related cost, not the provisional reserve. Expand the
|
|
1104
|
-
// recent suffix only while every related turn remains strictly older.
|
|
1105
|
-
const expanded = selectRecent(
|
|
1106
|
-
availableTokens - related.reduce((sum, turn) => sum + turn.tokens, 0),
|
|
1107
|
-
availableRows - related.reduce((sum, turn) => sum + turn.text.length, 0),
|
|
1108
|
-
);
|
|
1109
|
-
while (expanded.length > recent.length && related.some(turn => (
|
|
1110
|
-
!bucketBefore(turn, expanded[0]) || bucketOverlap(turn, expanded[0])
|
|
1111
|
-
))) expanded.shift();
|
|
1112
|
-
if (expanded.length > recent.length) recent = expanded;
|
|
1113
|
-
}
|
|
1114
1031
|
related.sort((a, b) => bucketBefore(a, b) ? -1 : bucketBefore(b, a) ? 1 : 0);
|
|
1115
1032
|
|
|
1116
1033
|
const relatedMessages = related.flatMap(turn => turn.text);
|
|
1117
|
-
const
|
|
1118
|
-
.filter(isVisibleConversationRow);
|
|
1119
|
-
const recentWasFitted = recent.some(turn => turn.text !== turn.messages
|
|
1120
|
-
&& (turn.text.length !== bucketTextMessages(turn.messages).length
|
|
1121
|
-
|| turn.tokens !== estimateMessagesTokens(bucketTextMessages(turn.messages))));
|
|
1122
|
-
const recentBaseline = recent.flatMap(turn => recentWasFitted ? turn.text : bucketTextMessages(turn.messages));
|
|
1034
|
+
const recentBaseline = recent.flatMap(turn => turn.text);
|
|
1123
1035
|
// Enrich only after both complete-text buckets and the active turn are paid.
|
|
1124
1036
|
// Tool protocol is useful only for immediate continuity; unlike visible text,
|
|
1125
1037
|
// it never reaches farther back than the configured recent tool window.
|
|
1126
|
-
const toolCutIndex = indexOfNthTurnFromEnd(recentText, keepToolTurns);
|
|
1127
1038
|
const recentToolSource = keepToolTurns > 0
|
|
1128
|
-
?
|
|
1039
|
+
? recent.slice(-keepToolTurns).flatMap(turn => turn.messages).filter(isVisibleConversationRow)
|
|
1129
1040
|
: [];
|
|
1130
1041
|
const recentMessages = withoutHistorySourceIndexes(addOptionalRecentToolPairs(
|
|
1131
1042
|
recentBaseline,
|
|
@@ -1154,8 +1065,12 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1154
1065
|
budget: {
|
|
1155
1066
|
messageTokenBudget: tokenBudget, maxMessageCount: messageCap,
|
|
1156
1067
|
recentTurnCap: recentCap, relatedTurnCap: relatedCap,
|
|
1157
|
-
|
|
1158
|
-
|
|
1068
|
+
minimumRecentTurns: 0,
|
|
1069
|
+
relatedReservedTokens: 0, availableHistoryTokens: availableTokens,
|
|
1070
|
+
usedTokens: estimateMessagesTokens([...relatedMessages, ...recentMessages]),
|
|
1071
|
+
usedMessages: relatedMessages.length + recentMessages.length,
|
|
1072
|
+
requestTokensBeforeWholeRequestFit: estimateMessagesTokens(messages),
|
|
1073
|
+
requestMessagesBeforeWholeRequestFit: messages.length,
|
|
1159
1074
|
},
|
|
1160
1075
|
dropped: {
|
|
1161
1076
|
pastTurnCount: droppedTurns.length,
|
|
@@ -1169,6 +1084,97 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1169
1084
|
};
|
|
1170
1085
|
}
|
|
1171
1086
|
|
|
1087
|
+
/**
|
|
1088
|
+
* Fit one provider-request copy to the actual model window. The caller tells
|
|
1089
|
+
* us where current-turn rows begin; only the prefix is subject to the history
|
|
1090
|
+
* budget. If the complete request is still too large, old history disappears
|
|
1091
|
+
* first, followed by the oldest disposable current-turn protocol units. The
|
|
1092
|
+
* source array and durable transcript are never mutated.
|
|
1093
|
+
*
|
|
1094
|
+
* @param {Array<object>} messages
|
|
1095
|
+
* @param {{ contextWindow:number, systemTokens?:number, toolSchemaTokens?:number,
|
|
1096
|
+
* outputReserve?:number, historyMessageCount?:number, historyTokenBudget?:number,
|
|
1097
|
+
* maxMessageCount?:number, language?:string }} options
|
|
1098
|
+
* @returns {{messages:Array<object>, meta:object}}
|
|
1099
|
+
*/
|
|
1100
|
+
export function fitProviderRequestToContext(messages, options = {}) {
|
|
1101
|
+
const source = Array.isArray(messages) ? messages : [];
|
|
1102
|
+
const contextWindow = bucketCap(options.contextWindow, 0);
|
|
1103
|
+
const staticTokens = bucketCap(options.systemTokens, 0)
|
|
1104
|
+
+ bucketCap(options.toolSchemaTokens, 0)
|
|
1105
|
+
+ bucketCap(options.outputReserve, 0);
|
|
1106
|
+
const messageBudget = Math.max(0, contextWindow - staticTokens);
|
|
1107
|
+
const split = Math.max(0, Math.min(source.length,
|
|
1108
|
+
Number.isInteger(options.historyMessageCount) ? options.historyMessageCount : 0));
|
|
1109
|
+
// The runtime cache's 256-row cap is a history-storage concern, not a model
|
|
1110
|
+
// request limit. Current-turn tool loops may legitimately exceed it while
|
|
1111
|
+
// remaining inside the model window. Only enforce a cap when the caller
|
|
1112
|
+
// explicitly supplies one.
|
|
1113
|
+
const messageCap = options.maxMessageCount === undefined
|
|
1114
|
+
? Number.MAX_SAFE_INTEGER
|
|
1115
|
+
: bucketCap(options.maxMessageCount, Number.MAX_SAFE_INTEGER);
|
|
1116
|
+
const historySource = source.slice(0, split);
|
|
1117
|
+
const currentSource = source.slice(split);
|
|
1118
|
+
|
|
1119
|
+
let current = pairSanitize(truncateToolResultsForModel(
|
|
1120
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
1121
|
+
));
|
|
1122
|
+
if (estimateMessagesTokens(current) > messageBudget || current.length > messageCap) {
|
|
1123
|
+
const fitted = [];
|
|
1124
|
+
if (current.length > 0 && messageBudget >= 2 && messageCap > 0) {
|
|
1125
|
+
const first = shrinkMessageToBudget(stripAllToolNoise([current[0]])[0], messageBudget);
|
|
1126
|
+
if (first && estimateMessageTokens(first) <= messageBudget) fitted.push(first);
|
|
1127
|
+
let tokens = messageBudget - estimateMessagesTokens(fitted);
|
|
1128
|
+
let rows = messageCap - fitted.length;
|
|
1129
|
+
const units = providerUnits(pairSanitize(current.slice(1)));
|
|
1130
|
+
const tail = [];
|
|
1131
|
+
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
1132
|
+
const unit = fitProviderUnit(units[index], tokens);
|
|
1133
|
+
const cost = estimateMessagesTokens(unit);
|
|
1134
|
+
if (unit.length > rows || cost > tokens) continue;
|
|
1135
|
+
tail.unshift(unit);
|
|
1136
|
+
tokens -= cost;
|
|
1137
|
+
rows -= unit.length;
|
|
1138
|
+
}
|
|
1139
|
+
fitted.push(...tail.flat());
|
|
1140
|
+
}
|
|
1141
|
+
current = pairSanitize(fitted);
|
|
1142
|
+
}
|
|
1143
|
+
|
|
1144
|
+
const configuredHistoryBudget = bucketCap(
|
|
1145
|
+
options.historyTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET,
|
|
1146
|
+
);
|
|
1147
|
+
const remainingTokens = Math.max(0, Math.min(
|
|
1148
|
+
configuredHistoryBudget,
|
|
1149
|
+
messageBudget - estimateMessagesTokens(current),
|
|
1150
|
+
));
|
|
1151
|
+
const remainingRows = Math.max(0, messageCap - current.length);
|
|
1152
|
+
const history = remainingTokens >= 2 && remainingRows > 0
|
|
1153
|
+
? trimSnapshotForBudget(historySource, {
|
|
1154
|
+
messageTokenBudget: remainingTokens,
|
|
1155
|
+
maxMessageCount: remainingRows,
|
|
1156
|
+
recentTurnCap: Number.MAX_SAFE_INTEGER,
|
|
1157
|
+
language: options.language,
|
|
1158
|
+
})
|
|
1159
|
+
: [];
|
|
1160
|
+
const fittedMessages = [...history, ...current];
|
|
1161
|
+
return {
|
|
1162
|
+
messages: fittedMessages,
|
|
1163
|
+
meta: {
|
|
1164
|
+
contextWindow,
|
|
1165
|
+
staticTokens,
|
|
1166
|
+
messageBudget,
|
|
1167
|
+
estimatedTokens: staticTokens + estimateMessagesTokens(fittedMessages),
|
|
1168
|
+
historyMessagesBefore: historySource.length,
|
|
1169
|
+
historyMessagesAfter: history.length,
|
|
1170
|
+
currentMessagesBefore: currentSource.length,
|
|
1171
|
+
currentMessagesAfter: current.length,
|
|
1172
|
+
droppedHistoryMessages: historySource.length - history.length,
|
|
1173
|
+
droppedCurrentMessages: currentSource.length - current.length,
|
|
1174
|
+
},
|
|
1175
|
+
};
|
|
1176
|
+
}
|
|
1177
|
+
|
|
1172
1178
|
/**
|
|
1173
1179
|
* Bound the Session-level runtime history cache. This is deliberately stricter
|
|
1174
1180
|
* than the provider configuration: the cache is only a disposable source
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Non-blocking, post-response conversation compaction.
|
|
3
|
+
*
|
|
4
|
+
* The compact artifact is a derived provider-context cache. It never replaces
|
|
5
|
+
* or tombstones ConversationStore rows. A generation fence in Engine decides
|
|
6
|
+
* whether a completed artifact is still current before this module writes it.
|
|
7
|
+
*/
|
|
8
|
+
import { promises as fs } from 'fs';
|
|
9
|
+
import { dirname, join } from 'path';
|
|
10
|
+
|
|
11
|
+
export const POST_COMPACT_CONTEXT_RATIO = 0.8;
|
|
12
|
+
|
|
13
|
+
function safePart(value, fallback) {
|
|
14
|
+
const text = typeof value === 'string' && value.trim() ? value.trim() : fallback;
|
|
15
|
+
return encodeURIComponent(text).replace(/%/g, '_');
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function postCompactPath(yeaftDir, { sessionId, vpId, threadId } = {}) {
|
|
19
|
+
if (!yeaftDir || !sessionId) return null;
|
|
20
|
+
const file = `${safePart(vpId, 'default')}--${safePart(threadId, 'main')}.json`;
|
|
21
|
+
return join(yeaftDir, 'sessions', safePart(sessionId, 'session'), 'conversation', 'post-compact', file);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export async function loadPostCompact(path) {
|
|
25
|
+
if (!path) return null;
|
|
26
|
+
try {
|
|
27
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
28
|
+
return parsed && parsed.version === 1 && typeof parsed.summary === 'string'
|
|
29
|
+
? parsed : null;
|
|
30
|
+
} catch {
|
|
31
|
+
return null;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export async function savePostCompact(path, artifact, isCurrent = null) {
|
|
36
|
+
if (!path) return false;
|
|
37
|
+
await fs.mkdir(dirname(path), { recursive: true });
|
|
38
|
+
const temp = `${path}.${process.pid}.${Date.now()}.tmp`;
|
|
39
|
+
await fs.writeFile(temp, `${JSON.stringify({ version: 1, ...artifact }, null, 2)}\n`, 'utf8');
|
|
40
|
+
if (typeof isCurrent === 'function' && !isCurrent()) {
|
|
41
|
+
await fs.unlink(temp).catch(() => {});
|
|
42
|
+
return false;
|
|
43
|
+
}
|
|
44
|
+
await fs.rename(temp, path);
|
|
45
|
+
return true;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export async function removePostCompactIfSource(path, sourceTurnId) {
|
|
49
|
+
if (!path || !sourceTurnId) return false;
|
|
50
|
+
try {
|
|
51
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
52
|
+
if (parsed?.sourceTurnId !== sourceTurnId) return false;
|
|
53
|
+
await fs.unlink(path);
|
|
54
|
+
return true;
|
|
55
|
+
} catch {
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export async function generatePostCompact({ adapter, model, messages, maxTokens = 4096 }) {
|
|
61
|
+
const transcript = JSON.stringify((Array.isArray(messages) ? messages : []).map(message => ({
|
|
62
|
+
role: message?.role,
|
|
63
|
+
content: message?.content,
|
|
64
|
+
...(Array.isArray(message?.toolCalls) ? { toolCalls: message.toolCalls } : {}),
|
|
65
|
+
...(message?.toolCallId ? { toolCallId: message.toolCallId } : {}),
|
|
66
|
+
})));
|
|
67
|
+
const result = await adapter.call({
|
|
68
|
+
model,
|
|
69
|
+
system: 'Summarize the earlier conversation for use as context in a later turn. Preserve user goals, decisions, constraints, unresolved work, and important results. Omit raw tool payloads and do not invent facts. Return only the compact summary.',
|
|
70
|
+
messages: [{ role: 'user', content: `Compact this transcript:\n${transcript}` }],
|
|
71
|
+
maxTokens,
|
|
72
|
+
});
|
|
73
|
+
const summary = typeof result?.text === 'string' ? result.text.trim() : '';
|
|
74
|
+
if (!summary) throw new Error('post compact returned empty content');
|
|
75
|
+
return summary;
|
|
76
|
+
}
|
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
* tool-folding/index.js — V7 reflection subsystem entry (PR-L).
|
|
3
3
|
*
|
|
4
4
|
* Exposes:
|
|
5
|
-
* - Constants
|
|
5
|
+
* - Constants TOOL_LOOP_REFLECTION_INTERVAL, TURN_SUMMARY_THRESHOLD,
|
|
6
|
+
* DUP_TOOL_THRESHOLD
|
|
6
7
|
* - Reflector helpers (T1 sync, T2 async, fallback stub)
|
|
7
8
|
* - Helpers for collapsing message ranges into a single assistant
|
|
8
9
|
* reflection message
|
|
@@ -10,19 +11,10 @@
|
|
|
10
11
|
*
|
|
11
12
|
* The constants are NOT config-driven — V7 design freezes them in code.
|
|
12
13
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* end-of-turn reflection path. Keep a usefully wide gap between the two
|
|
18
|
-
* so the (T2, T1) band where T2-alone applies stays meaningful.
|
|
19
|
-
*
|
|
20
|
-
* TOOL_BATCH_SIZE history: was 13 originally; raised to 30 (2026-05-15)
|
|
21
|
-
* after user feedback that 13 fired too often inside a single task and
|
|
22
|
-
* fragmented otherwise-coherent tool arcs into multiple reflections. 30
|
|
23
|
-
* keeps the periodic-reflection contract (it still fires every N tools,
|
|
24
|
-
* not just once) but gives a single task arc room to breathe before the
|
|
25
|
-
* arc gets collapsed.
|
|
14
|
+
* T1 runs inside the turn and collapses history in place. Its cadence is
|
|
15
|
+
* measured in provider tool loops (assistant tool_use batch → execution →
|
|
16
|
+
* next provider boundary), not in the number of calls inside a batch. A model
|
|
17
|
+
* returning 30 parallel tools has completed one loop, not thirty.
|
|
26
18
|
*
|
|
27
19
|
* TURN_SUMMARY_THRESHOLD history: was 5 originally; raised to 8
|
|
28
20
|
* (2026-05-18). 5 was too aggressive — small "read a few files, edit one,
|
|
@@ -32,7 +24,9 @@
|
|
|
32
24
|
* before the next turn's history grows.
|
|
33
25
|
*/
|
|
34
26
|
|
|
35
|
-
export const
|
|
27
|
+
export const TOOL_LOOP_REFLECTION_INTERVAL = 30;
|
|
28
|
+
// Compatibility for external imports; the engine uses the loop-specific name.
|
|
29
|
+
export const TOOL_BATCH_SIZE = TOOL_LOOP_REFLECTION_INTERVAL;
|
|
36
30
|
export const TURN_SUMMARY_THRESHOLD = 8;
|
|
37
31
|
export const DUP_TOOL_THRESHOLD = 3;
|
|
38
32
|
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* t1-reflector.js — V7 in-turn (synchronous) reflection (PR-L).
|
|
3
3
|
*
|
|
4
|
-
* Triggered
|
|
5
|
-
*
|
|
4
|
+
* Triggered after each interval of 30 completed tool loops, immediately before
|
|
5
|
+
* the engine loops back into adapter.stream(). Parallel calls returned in one
|
|
6
|
+
* assistant tool-use batch count as one loop. Calls
|
|
6
7
|
* the PRIMARY model — never the fast model — to generate a markdown
|
|
7
8
|
* reflection over the batch.
|
|
8
9
|
*
|