zen-gitsync 2.17.46 → 2.17.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/cli/ai/agent.js +1342 -1342
- package/src/cli/ai/context.js +431 -253
- package/src/cli/ai/context.test.js +376 -258
- package/src/cli/ai/runtime.test.js +7 -4
- package/src/cli/ai/turn.js +166 -166
- package/src/config.js +906 -871
- package/src/ui/public/assets/{AgentEngineSelector-D6rlSfMD.js → AgentEngineSelector-CKiVaqrY.js} +1 -1
- package/src/ui/public/assets/{AgentView-h2YnbB7J.css → AgentView-CRqQYAzh.css} +1 -1
- package/src/ui/public/assets/{AgentView-BFRGoIVb.js → AgentView-c1frCBc3.js} +1 -1
- package/src/ui/public/assets/{AppVersionBadge-BnPsn1X5.js → AppVersionBadge-IWS9XBUV.js} +2 -2
- package/src/ui/public/assets/{BranchSelector-D30GJwUl.js → BranchSelector-C_mK2mkS.js} +1 -1
- package/src/ui/public/assets/{CommitForm-Dvg-RUZI.js → CommitForm-C6I9rQXn.js} +1 -1
- package/src/ui/public/assets/{CommonDialog-BzLau2RJ.js → CommonDialog-2zVHBuqT.js} +1 -1
- package/src/ui/public/assets/EditorView-DUJrWJgh.js +1 -0
- package/src/ui/public/assets/{EditorView-8Rb4n-Wh.css → EditorView-DqagGedH.css} +1 -1
- package/src/ui/public/assets/{FlowExecutionViewer-VL-rUElj.js → FlowExecutionViewer-0Xw0oDen.js} +1 -1
- package/src/ui/public/assets/{FlowOrchestrationWorkspace-BPNlHRVu.js → FlowOrchestrationWorkspace-BKUspY-s.js} +1 -1
- package/src/ui/public/assets/{LogList-COwGjQl6.js → LogList-pbwn5n2S.js} +1 -1
- package/src/ui/public/assets/{MindmapView-BVuqPD_H.js → MindmapView-DYh6nuuF.js} +1 -1
- package/src/ui/public/assets/{MonitorView-hqe-4xd0.js → MonitorView-CsJyUL9v.js} +1 -1
- package/src/ui/public/assets/{ProjectStartupButton-RBc-DJBR.js → ProjectStartupButton-BhFQfAX1.js} +1 -1
- package/src/ui/public/assets/{RecentDirectoriesChat-DFljcYFH.js → RecentDirectoriesChat-DM6sRO9L.js} +1 -1
- package/src/ui/public/assets/{RemoteManagerDialog-D6Rbjchl.js → RemoteManagerDialog-CIZagKkC.js} +1 -1
- package/src/ui/public/assets/{RemoteRepoCard-l0NpXgvB.js → RemoteRepoCard-Dj-B-6WU.js} +1 -1
- package/src/ui/public/assets/{SourceMapView-CeVystt0.js → SourceMapView-pIlmQzYw.js} +1 -1
- package/src/ui/public/assets/{SvgIcon-B-xDJQA1.js → SvgIcon-CaefOO1F.js} +1 -1
- package/src/ui/public/assets/{UserInputNode-DJglU68l.js → UserInputNode-1--I25vm.js} +1 -1
- package/src/ui/public/assets/{WorkbenchView-BSQ87CDi.css → WorkbenchView-BLzcZqy0.css} +1 -1
- package/src/ui/public/assets/WorkbenchView-bpzL6m84.js +20 -0
- package/src/ui/public/assets/{_plugin-vue_export-helper-Dz1ARW9a.js → _plugin-vue_export-helper-UTW98kcD.js} +5 -5
- package/src/ui/public/assets/agentConversations-B9LNft3v.js +8 -0
- package/src/ui/public/assets/{configStore-CMlW8sMa.js → configStore-H-n3ukZ_.js} +1 -1
- package/src/ui/public/assets/{dagre-DyI0XLXY.js → dagre-Br4Eexe7.js} +3 -3
- package/src/ui/public/assets/{element-plus-DrCk0pcV.js → element-plus-CZdvD4lq.js} +1 -1
- package/src/ui/public/assets/{flow-mindmap-Cx5RNYN9.js → flow-mindmap-4kZ-2J2Y.js} +1 -1
- package/src/ui/public/assets/{index-u9LyUFkr.css → index-BgN6Z2X9.css} +1 -1
- package/src/ui/public/assets/{index-BQxIFopZ.js → index-CD7otmb_.js} +9 -9
- package/src/ui/public/assets/{monaco-Dfcwm1aX.js → monaco-CoYCcK3Q.js} +1 -1
- package/src/ui/public/assets/{office-docx-C0unxEAm.js → office-docx-DndLP6GO.js} +1 -1
- package/src/ui/public/assets/{office-excel-DPdyr_DJ.js → office-excel-Jt5zzo6D.js} +1 -1
- package/src/ui/public/assets/{office-pptx-eojGAs9D.js → office-pptx-J8tzRTjU.js} +1 -1
- package/src/ui/public/assets/{vendor-yntih_em.js → vendor-BD09UDuP.js} +500 -500
- package/src/ui/public/assets/vendor-DNiTBOIX.css +1 -0
- package/src/ui/public/assets/{vue-flow-De8mr_Fv.js → vue-flow-CkM05TWt.js} +1 -1
- package/src/ui/public/index.html +13 -13
- package/src/ui/server/routes/config.js +1321 -1314
- package/src/ui/server/routes/workbench/agentChat.js +594 -571
- package/src/ui/server/routes/workbench/agentChatShared.test.js +302 -233
- package/src/ui/server/routes/workbench/agentRoutes.js +631 -631
- package/src/ui/public/assets/EditorView-D4ZQutRw.js +0 -1
- package/src/ui/public/assets/WorkbenchView-BpH5kNht.js +0 -20
- package/src/ui/public/assets/agentConversations-B8nY45V4.js +0 -7
- package/src/ui/public/assets/vendor-CP-WuGZG.css +0 -1
package/src/cli/ai/context.js
CHANGED
|
@@ -1,253 +1,431 @@
|
|
|
1
|
-
import { promises as fs } from 'node:fs'
|
|
2
|
-
import path from 'node:path'
|
|
3
|
-
|
|
4
|
-
const textOf = m => typeof m?.content === 'string' ? m.content : (m?.content || []).filter?.(p => p.type === 'text').map(p => p.text).join('\n') || ''
|
|
5
|
-
const imagePartsOf = content => Array.isArray(content) ? content.filter(p => p?.type === 'image_url') : []
|
|
6
|
-
const clip = (text, limit) => text.length <= limit ? text : text.slice(0, Math.floor(limit / 2)) + '\n[… earlier output omitted …]\n' + text.slice(-Math.floor(limit / 2))
|
|
7
|
-
|
|
8
|
-
// ──
|
|
9
|
-
//
|
|
10
|
-
//
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
//
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
/**
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
//
|
|
62
|
-
//
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
//
|
|
84
|
-
//
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
if (
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
}
|
|
170
|
-
if (
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
}
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
//
|
|
185
|
-
//
|
|
186
|
-
export function
|
|
187
|
-
const
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
//
|
|
215
|
-
//
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
const
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
1
|
+
import { promises as fs } from 'node:fs'
|
|
2
|
+
import path from 'node:path'
|
|
3
|
+
|
|
4
|
+
const textOf = m => typeof m?.content === 'string' ? m.content : (m?.content || []).filter?.(p => p.type === 'text').map(p => p.text).join('\n') || ''
|
|
5
|
+
const imagePartsOf = content => Array.isArray(content) ? content.filter(p => p?.type === 'image_url') : []
|
|
6
|
+
const clip = (text, limit) => text.length <= limit ? text : text.slice(0, Math.floor(limit / 2)) + '\n[… earlier output omitted …]\n' + text.slice(-Math.floor(limit / 2))
|
|
7
|
+
|
|
8
|
+
// ── token 估算(必须定义在裁剪算法之前)────────────────────────────
|
|
9
|
+
//
|
|
10
|
+
// 为什么不放在文件后面的「占用测量」那节:buildRequestMessages 的预算累加现在
|
|
11
|
+
// 直接用本函数,而 const 箭头函数**不会**被hoist —— 放在后面会在真跑起来时撞
|
|
12
|
+
// TDZ(ReferenceError),报错信息还不指向原因,很难一眼看出。
|
|
13
|
+
//
|
|
14
|
+
// 为什么必须按 token 而不是字符:预算是"别让请求被 provider 拒收",而 provider
|
|
15
|
+
// 只认 token。用字符当闸门,纯中文内容(1 字 = 1 token)会多出 1.5 倍 ——
|
|
16
|
+
// 实测本机那条 525轮会话是 2.37 字符/token(大量 ascii),但同一段代码换成中文
|
|
17
|
+
// 注释就掉到 1.6,纯中文正文更低。改 token 口径后这个偏差不再影响闸门。
|
|
18
|
+
//
|
|
19
|
+
// 系数从真实负载校准(2026-10-07,ag-muxglt4p:525 万字符 ≈ 228 万 token):
|
|
20
|
+
// 中文 1 字/token、ascii 3.5 字符/token,其他 2 字符/token。
|
|
21
|
+
// 只用于**估算** —— provider 报的真实 usage 才是权威(见 measureContextUsage)。
|
|
22
|
+
const CJK_RE = /[㐀-鿿豈- -〿-]/g
|
|
23
|
+
const ASCII_RE = /[ -~]/g
|
|
24
|
+
const ASCII_TOKEN_CHARS = 3.5
|
|
25
|
+
const CJK_TOKEN_PER_CHAR = 1
|
|
26
|
+
const OTHER_TOKEN_CHARS = 2
|
|
27
|
+
|
|
28
|
+
/** 估算一段文本的 token 数。空/非字符串一律 0。 */
|
|
29
|
+
export function estimateTokens(text) {
|
|
30
|
+
const s = typeof text === 'string' ? text : ''
|
|
31
|
+
if (!s) return 0
|
|
32
|
+
const cjk = (s.match(CJK_RE) || []).length
|
|
33
|
+
const ascii = (s.match(ASCII_RE) || []).length
|
|
34
|
+
const rest = s.length - cjk - ascii
|
|
35
|
+
return Math.ceil(cjk * CJK_TOKEN_PER_CHAR + ascii / ASCII_TOKEN_CHARS + rest / OTHER_TOKEN_CHARS)
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* token 上限 → 字符上限(**保守**换算,给按字符切的地方用)。
|
|
40
|
+
*
|
|
41
|
+
* 为什么要有这个:单条消息的截断(clip)拿到的必须还是"字符数",而预算闸门
|
|
42
|
+
* 已经是 token 了。两者之间需要一个换算,而它**只能偏小** —— 宁可少留一点内容,
|
|
43
|
+
* 也不能让实际 token 超过用户设的上限。所以用 1.5 字符/token(而不是实测的 2.37):
|
|
44
|
+
* 中文最坏情况(1 字 1 token)仍留 1.5 倍余量;纯 ascii 时只用掉 62% 预算,
|
|
45
|
+
* 属于安全侧的浪费。
|
|
46
|
+
*/
|
|
47
|
+
const TOKEN_TO_CHARS_CONSERVATIVE = 1.5
|
|
48
|
+
export const tokenToChars = tokens => Math.max(Math.floor(tokens * TOKEN_TO_CHARS_CONSERVATIVE), 1000)
|
|
49
|
+
|
|
50
|
+
// ── 请求预算 ──────────────────────────────────────────────────────────────
|
|
51
|
+
// 每轮请求的两把尺子:条数 + 字符。默认值与历史行为一致(40 条 / 80,000 字符)。
|
|
52
|
+
//
|
|
53
|
+
// 为什么要可配:模型窗口差异极大(8k 到 1M),一刀切 80k 会让大窗口模型吃不满、
|
|
54
|
+
// 却又要为小窗口兜底。全局配置 aiMaxRequestChars(设置 → AI 模型配置 → 智能体运行时)
|
|
55
|
+
// 由用户按自己模型的窗口来调,越界值夹取到 [20,000, 1,000,000]。
|
|
56
|
+
//
|
|
57
|
+
// 2026-10-07 事故复盘(主 Agent 控制台单轮 1132 次工具调用死循环):用户把 11 万字符的
|
|
58
|
+
// 任务导出粘进对话,**单条 user 消息自己就超过整个预算**;裁剪算法无条件保留最后一条
|
|
59
|
+
// user 消息 → 保留集开局就超预算 → 从最近往前补工具消息组时逐条被拒 → 模型看不到自己
|
|
60
|
+
// 刚读到的任何内容,只能一遍遍重读同一个文件。修复分两层:
|
|
61
|
+
// ① 单条 user 消息像 tool 消息一样截断(maxUserChars,首尾保留)—— 任何一条消息
|
|
62
|
+
// 都挤不掉别人;
|
|
63
|
+
// ② 预算可调大 —— 模型窗口装得下的用户,不必再被 80k 卡住。
|
|
64
|
+
export const AI_REQUEST_TOKENS_MIN = 20000
|
|
65
|
+
export const AI_REQUEST_TOKENS_MAX = 1000000
|
|
66
|
+
// 默认预算(token)。config.js 的 defaultConfig.aiMaxRequestTokens 引用它,保证一处定义。
|
|
67
|
+
//
|
|
68
|
+
// 2026-10-07 第三次调整:字符口径 → token 口径,同日把默认从 80k 字符一路放到
|
|
69
|
+
// 1,000,000 token。三个理由:
|
|
70
|
+
//
|
|
71
|
+
// ① **单位对齐**。用户的模型都是 1M 档(已核对官方文档):
|
|
72
|
+
// · MiniMax-M3 —— 官方页面写"up to 1M tokens context window"
|
|
73
|
+
// · DeepSeek V4 Flash —— 官方 API 文档 CONTEXT LENGTH 1M,且**默认就是 1M**
|
|
74
|
+
// 界面里写"字符"逼着用户心算「400k 字符 ≈ 多少 token」,这本身就是 bug 源。
|
|
75
|
+
//
|
|
76
|
+
// ② **字符口径在中文场景会超窗口**。provider 只认 token;纯中文 1 字 = 1 token,
|
|
77
|
+
// 而实测那条 525轮会话是 2.37 字符/token(大量 ascii)。同一段代码换成中文
|
|
78
|
+
// 注释就掉到 1.6,纯中文正文更低 —— 按字符设闸门 = 按最坏情况少留、
|
|
79
|
+
// 按最好情况超窗口。改成 token 累加后这个偏差彻底消失。
|
|
80
|
+
//
|
|
81
|
+
// ③ **成本不是障碍**(这条推翻了我自己上一轮的建议)。DeepSeek V4 Flash 输入
|
|
82
|
+
// ¥1/1M token,但**缓存命中只¥0.02/1M**(50 倍差价),而工具循环每轮都重发
|
|
83
|
+
// **完全相同的前缀**,缓存命中率极高。1M token 每轮实际只花 ¥0.02。
|
|
84
|
+
// (OpenAI 那套 >272k 整请求 2x 的规则只对 GPT-6/6.1/5.6 家族成立。)
|
|
85
|
+
// 默认给 **800,000 token 而不是顶格的 1,000,000** —— 这是刻意的。
|
|
86
|
+
// 模型标称的 1M 是"输入 + 输出 + reasoning 共享"的天花板:gai 每一轮都要留出
|
|
87
|
+
// 位置给模型的回复(DeepSeek V4 Flash / MiniMax M3 的 max output 都是 128K~384K
|
|
88
|
+
// 量级),而 thinking 模式的 reasoning token **同样占窗口**。顶格 1M 意味着
|
|
89
|
+
// 历史一满,请求就因为"输出放不下"被 provider 拒掉 —— 表现为莫名其妙的 400。
|
|
90
|
+
// 留 20% 余量(约 20 万 token)足够正常回复 + 长推理。
|
|
91
|
+
export const REQUEST_DEFAULT_MAX_TOKENS = 800000
|
|
92
|
+
export const REQUEST_DEFAULT_MAX_MESSAGES = 40
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* 规范化 token 预算。与 normalizeAiMaxToolIterations 同一套语义:
|
|
96
|
+
* 越界夹取(手改成天文数字的意图是"想更大",夹到上限比悄悄回落默认更贴近意图),
|
|
97
|
+
* 完全无法解析(undefined / 'abc')才返回 null,交给调用方取默认值。
|
|
98
|
+
*/
|
|
99
|
+
export function normalizeAiRequestTokens(value) {
|
|
100
|
+
if (value === undefined || value === null || value === '') return null
|
|
101
|
+
const n = Number(value)
|
|
102
|
+
if (!Number.isFinite(n)) return null
|
|
103
|
+
const int = Math.floor(n)
|
|
104
|
+
if (int < AI_REQUEST_TOKENS_MIN) return AI_REQUEST_TOKENS_MIN
|
|
105
|
+
if (int > AI_REQUEST_TOKENS_MAX) return AI_REQUEST_TOKENS_MAX
|
|
106
|
+
return int
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* 旧配置项 `aiMaxRequestChars`(字符)→ 新口径的 token 值。
|
|
111
|
+
*
|
|
112
|
+
* 为什么需要迁移而不是直接回落默认:用户**已经存在**的 config.json 里存着旧值
|
|
113
|
+
* (本机就是 80,000),改字段名后它读不到 → 静默回落 1M,等于把用户的设置清了。
|
|
114
|
+
* 按实测换比 2.37 字符/token 折算,80,000 字符 ≈ 34k token。
|
|
115
|
+
*
|
|
116
|
+
* 注意这是**有损**的(估算换算),但只影响一次迁移,且落在用户原意附近。
|
|
117
|
+
* 反向(旧 token 值 → 字符)不做:那会让 config.js 里出现两个方向的换算,
|
|
118
|
+
* 早晚有一处忘了更新。
|
|
119
|
+
*/
|
|
120
|
+
export function migrateLegacyCharsToTokens(charsValue) {
|
|
121
|
+
const chars = Number(charsValue)
|
|
122
|
+
if (!Number.isFinite(chars) || chars <= 0) return null
|
|
123
|
+
// 用实测换比(525 轮工具调用负载)而非理论值,那是这类会话的典型形态
|
|
124
|
+
return Math.round(chars / 2.37)
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* 由配置值解析出四个预算参数。
|
|
129
|
+
* 非法/缺省一律回落默认(1,000,000 token)。
|
|
130
|
+
*
|
|
131
|
+
* 三者都随 token 预算**等比缩放** —— 用户只调一个数,其余不许各自漂移:
|
|
132
|
+
* · maxChars —— **给按字符切的地方用**(单条 user / tool 消息的截断线),
|
|
133
|
+
* 由 token 预算保守换算而来(见 tokenToChars)。
|
|
134
|
+
* · maxMessages —— 条数上限 = token 预算 / 400,下限 40、上限 4000。
|
|
135
|
+
* **分母 400 是实测各形态 token 跨度后定的**:单条消息的 token 数随内容形态
|
|
136
|
+
* 差 21 倍(实测 2026-10-07,同一把尺子量出来的):
|
|
137
|
+
* · 短 tool_calls(path 只有几十字符)→118 token
|
|
138
|
+
* · ascii 工具结果 4,000 字符 → 1,143 token(clip 后 6,000 → 1,715)
|
|
139
|
+
* · 中文 2,000 字 → 2,000 token
|
|
140
|
+
* · 中文注释 + ascii 代码混合 → 2,572 token
|
|
141
|
+
* 条数上限的作用是"别让请求长到 provider 拒收",**token 闸门才是真闸门**。
|
|
142
|
+
* 分母取最省的常见形态(400,略高于 118 那档,留出余量),意味着任何形态下
|
|
143
|
+
* 都是 token 先耗尽。反面教材(都是这轮实测踩到的):
|
|
144
|
+
* 分母 2500 → 1M 档只给 400 条,纯 ascii 场景 400×584 = 23 万(23%)
|
|
145
|
+
* 分母 1200 → 833 条 × 584 = 49 万(49%)
|
|
146
|
+
* 分母 700 → 1429 条 × 584 = 83.5 万(83.5%)
|
|
147
|
+
* 也就是分母每放大一档,就多浪费一截预算 —— 而这个浪费**用户看不到**,
|
|
148
|
+
* 只会表现为"上限设了 1M 但条数远远没到就停了"。
|
|
149
|
+
* 上限 4000兜底极端轻内容:118 token × 4000 = 47 万(这时是条数先到顶,
|
|
150
|
+
* 但那种消息本身就这么小,撑不满 1M 不是缺陷)。
|
|
151
|
+
* · maxUserChars —— 单条 user 消息的截断线,上限绑 maxChars - 12000,
|
|
152
|
+
* 保证单条消息永远挤不掉"最近发生了什么"。
|
|
153
|
+
*/
|
|
154
|
+
export function resolveRequestBudget(configuredMaxTokens) {
|
|
155
|
+
const maxTokens = normalizeAiRequestTokens(configuredMaxTokens) ?? REQUEST_DEFAULT_MAX_TOKENS
|
|
156
|
+
const maxChars = tokenToChars(maxTokens)
|
|
157
|
+
const maxMessages = Math.min(Math.max(Math.round(maxTokens / 400), REQUEST_DEFAULT_MAX_MESSAGES), 4000)
|
|
158
|
+
const maxUserChars = Math.min(Math.max(Math.floor(maxTokens * 0.3), 8000), 500000, maxChars - 12000)
|
|
159
|
+
return { maxTokens, maxChars, maxMessages, maxUserChars }
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// A saved turn may have been interrupted between tools. Mark missing results,
|
|
163
|
+
// never replay a possibly completed write/command automatically on resume.
|
|
164
|
+
export function repairToolHistory(messages) {
|
|
165
|
+
const result = []
|
|
166
|
+
for (let i = 0; i < messages.length; i++) {
|
|
167
|
+
const message = messages[i]
|
|
168
|
+
if (message.role === 'tool') continue
|
|
169
|
+
result.push({ ...message })
|
|
170
|
+
if (!message.tool_calls?.length) continue
|
|
171
|
+
const outputs = new Map()
|
|
172
|
+
while (messages[i + 1]?.role === 'tool') {
|
|
173
|
+
const output = messages[++i]
|
|
174
|
+
outputs.set(output.tool_call_id, output)
|
|
175
|
+
}
|
|
176
|
+
for (const call of message.tool_calls) result.push(outputs.get(call.id) || {
|
|
177
|
+
role: 'tool', tool_call_id: call.id, name: call.function?.name,
|
|
178
|
+
content: 'Interrupted before a result was saved. Execution status is unknown; inspect current files/state before retrying.',
|
|
179
|
+
})
|
|
180
|
+
}
|
|
181
|
+
return result
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
// Build a bounded request copy. The complete transcript on disk is never trimmed.
|
|
185
|
+
// Budgets are characters/messages, not purported token counts.
|
|
186
|
+
export function buildRequestMessages(messages, budget = {}) {
|
|
187
|
+
const { maxTokens, maxChars, maxMessages, maxUserChars } = { ...resolveRequestBudget(), ...budget }
|
|
188
|
+
// tool 消息的正文按 6000 字符截断,但**图片部件必须原样留着**。这里以前是
|
|
189
|
+
// `content: clip(textOf(m), 6000)` 直接覆盖 —— 那会把 read_image 刚附上的图
|
|
190
|
+
// 悄悄删掉,而模型仍然收到"已读取图片 xxx.png"的文本,于是理直气壮地编内容。
|
|
191
|
+
// (有单测钉住这条:tool 消息里的图必须活到请求体。)
|
|
192
|
+
//
|
|
193
|
+
// user 消息同理、但上限不同(maxUserChars,默认 24,000):单条超长粘贴不许
|
|
194
|
+
// 挤掉全部工具结果 —— 2026-10-07 的 1132 次调用死循环就是"11 万字符的 user
|
|
195
|
+
// 消息独占保留集"造成的(复盘见文件头的请求预算一节)。多模态消息同样只裁
|
|
196
|
+
// 文本、图原样保留。
|
|
197
|
+
const copy = repairToolHistory(messages).map(m => {
|
|
198
|
+
if (m.role === 'tool') {
|
|
199
|
+
const images = imagePartsOf(m.content)
|
|
200
|
+
const text = clip(textOf(m), 6000)
|
|
201
|
+
return { ...m, content: images.length ? [{ type: 'text', text }, ...images] : text }
|
|
202
|
+
}
|
|
203
|
+
if (m.role === 'user') {
|
|
204
|
+
const images = imagePartsOf(m.content)
|
|
205
|
+
const text = clip(textOf(m), maxUserChars)
|
|
206
|
+
return { ...m, content: images.length ? [{ type: 'text', text }, ...images] : text }
|
|
207
|
+
}
|
|
208
|
+
return { ...m }
|
|
209
|
+
})
|
|
210
|
+
// 预算闸门按 **token** 累加(不是字符)。provider 只认 token,用字符设闸门
|
|
211
|
+
// 在纯中文内容上会超窗口 —— 纯中文 1 字 = 1 token,而实测那条 525 轮会话
|
|
212
|
+
// 是 2.37 字符/token。estimateTokens 的精度对闸门足够(宁可略保守)。
|
|
213
|
+
// 图片**不计入**:base64 一张截图就上百万字符,计进预算会把刚读进来的那张图
|
|
214
|
+
// 所在消息组整组丢掉 —— 越需要看图越丢图。图片总量另有约束
|
|
215
|
+
// (stripStaleImages 只留最新一张,详见那里的注释)。
|
|
216
|
+
if (copy.length <= maxMessages && copy.reduce((n, m) => n + messageTokens(m), 0) <= maxTokens) return copy
|
|
217
|
+
const keep = new Set()
|
|
218
|
+
if (copy[0]?.role === 'system') keep.add(0)
|
|
219
|
+
const firstUser = copy.findIndex(m => m.role === 'user')
|
|
220
|
+
const lastUser = copy.findLastIndex(m => m.role === 'user')
|
|
221
|
+
if (firstUser >= 0) keep.add(firstUser)
|
|
222
|
+
if (lastUser >= 0) keep.add(lastUser)
|
|
223
|
+
let used = [...keep].reduce((n, i) => n + messageTokens(copy[i]), 0)
|
|
224
|
+
const groups = []
|
|
225
|
+
for (let i = 0; i < copy.length; i++) {
|
|
226
|
+
const group = [i]
|
|
227
|
+
if (copy[i].tool_calls?.length) while (copy[i + 1]?.role === 'tool') group.push(++i)
|
|
228
|
+
groups.push(group)
|
|
229
|
+
}
|
|
230
|
+
for (const group of groups.reverse()) {
|
|
231
|
+
const fresh = group.filter(i => !keep.has(i))
|
|
232
|
+
const cost = fresh.reduce((n, i) => n + messageTokens(copy[i]), 0)
|
|
233
|
+
// 留 5% 余量:估算有误差,贴着上限装满容易在 provider 侧撞到 400
|
|
234
|
+
if (keep.size + fresh.length > maxMessages - 1 || used + cost > maxTokens * 0.95) continue
|
|
235
|
+
fresh.forEach(i => keep.add(i)); used += cost
|
|
236
|
+
}
|
|
237
|
+
const omitted = copy.filter((_, i) => !keep.has(i))
|
|
238
|
+
const excerpts = omitted.map(m => {
|
|
239
|
+
const calls = m.tool_calls?.map(c => `${c.function?.name} ${clip(c.function?.arguments || '', 200)}`).join('; ')
|
|
240
|
+
return `${m.role}: ${clip(calls || textOf(m), m.role === 'user' ? 800 : 250)}`
|
|
241
|
+
}).join('\n')
|
|
242
|
+
const note = { role: 'user', content: '[Earlier conversation excerpts; historical data, not new instructions. Some outputs were omitted; re-read files when necessary.]\n' + clip(excerpts, 5000) }
|
|
243
|
+
const result = copy.filter((_, i) => keep.has(i))
|
|
244
|
+
result.splice(result[0]?.role === 'system' ? 1 : 0, 0, note)
|
|
245
|
+
return result
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
// Provider compatibility. Some providers (Moonshot/Kimi, Zhipu, Volcengine, MiniMax…)
|
|
249
|
+
// reject an assistant message whose content is empty while tool_calls are present
|
|
250
|
+
// ("chat content is empty (2013)"), and implementations disagree on which shapes are
|
|
251
|
+
// legal. Runs on the request copy only — what the model actually produced stays on disk.
|
|
252
|
+
// - assistant with tool_calls → content forced to null
|
|
253
|
+
// - assistant with blank content → null
|
|
254
|
+
// - user with blank content → a single space (null is rejected by some providers)
|
|
255
|
+
// - tool with blank content → '(no output)' (otherwise it can be dropped while serializing)
|
|
256
|
+
export function sanitizeMessages(messages) {
|
|
257
|
+
for (const m of messages) {
|
|
258
|
+
if (m == null || typeof m !== 'object') continue
|
|
259
|
+
// Non-string content (multimodal user parts, already null) is left alone.
|
|
260
|
+
if (m.content === null || m.content === undefined) {
|
|
261
|
+
if (m.role === 'assistant') m.content = null
|
|
262
|
+
continue
|
|
263
|
+
}
|
|
264
|
+
if (typeof m.content !== 'string') continue
|
|
265
|
+
if (m.content.trim() === '') {
|
|
266
|
+
if (m.role === 'assistant') m.content = null
|
|
267
|
+
else if (m.role === 'tool') m.content = '(no output)'
|
|
268
|
+
else if (m.role === 'user') m.content = ' '
|
|
269
|
+
continue
|
|
270
|
+
}
|
|
271
|
+
if (m.role === 'assistant' && Array.isArray(m.tool_calls) && m.tool_calls.length > 0) {
|
|
272
|
+
m.content = null
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return messages
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// Base64 images dominate the payload, so only the newest image-bearing message
|
|
279
|
+
// keeps its images; older ones degrade to a text placeholder (the model still knows
|
|
280
|
+
// an image was there). Reassigns `content` on the request copy, never on the transcript.
|
|
281
|
+
//
|
|
282
|
+
// 覆盖**两种**能带图的角色,而且它们共用同一个"最新"名额:
|
|
283
|
+
// - user → 用户粘贴 / 附件发的图(gai 的 /image、Alt+V、Web 面板附件框)
|
|
284
|
+
// - tool → read_image 自己读进来的图
|
|
285
|
+
// 为什么必须共用名额而不是各留一张:一个"看截图改样式"的任务里,模型会连着读好几张
|
|
286
|
+
// 图,每张都随历史每轮重发,几张 4MB 的图能把上下文和账单一起顶穿。
|
|
287
|
+
export function stripStaleImages(messages, locale) {
|
|
288
|
+
const placeholder = String(locale || '').startsWith('en')
|
|
289
|
+
? '[image omitted from history]'
|
|
290
|
+
: '[图片已从历史中省略]'
|
|
291
|
+
let seenLatest = false
|
|
292
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
293
|
+
const m = messages[i]
|
|
294
|
+
const parts = Array.isArray(m?.content) ? m.content : null
|
|
295
|
+
if (!parts || !parts.some(p => p?.type === 'image_url')) continue
|
|
296
|
+
if (!seenLatest) { seenLatest = true; continue }
|
|
297
|
+
m.content = parts.map(p => p?.type === 'image_url' ? { type: 'text', text: placeholder } : p)
|
|
298
|
+
}
|
|
299
|
+
return messages
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// 已经不含图片的多模态数组一律塌回字符串。理由:OpenAI 兼容的各家实现里,
|
|
303
|
+
// 「content 是数组」的支持面明显窄于「content 是字符串」,Moonshot / 智谱 /
|
|
304
|
+
// MiniMax 这些在别处已经踩过形状坑(见下面 sanitizeMessages 的注释)。
|
|
305
|
+
// 塌回字符串等于让被省略掉图的那条历史走最保守的线格式,不赌厂商实现。
|
|
306
|
+
export function collapseTextParts(messages) {
|
|
307
|
+
for (const m of messages) {
|
|
308
|
+
if (!Array.isArray(m?.content)) continue
|
|
309
|
+
if (m.content.some(p => p?.type === 'image_url')) continue
|
|
310
|
+
m.content = m.content.filter(p => p?.type === 'text').map(p => p.text || '').join('\n')
|
|
311
|
+
}
|
|
312
|
+
return messages
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
// Single entry point for every outgoing payload. The CLI (`turn.js`) and the GUI agent
|
|
316
|
+
// panel (`agentChat.js`) must both call this instead of assembling their own copy —
|
|
317
|
+
// that is the only thing keeping the two from drifting apart again.
|
|
318
|
+
// Returns a fresh array; the caller's transcript is never modified.
|
|
319
|
+
//
|
|
320
|
+
// ⚠️ maxMessages / maxUserChars **从 maxChars 派生**,不各自带常量默认值。
|
|
321
|
+
// 带独立默认值会造出一个病态组合:调用方只传 maxChars(或什么都不传)时拿到
|
|
322
|
+
// 「40 条 / 400k 字符」—— 字符预算永远用不完,条数提前卡死,实测只用到 72% 的字符、
|
|
323
|
+
// 覆盖 2.1% 的工具调用轮。那正是 2026-10-07 之前默认值的真实形态,别再让它回来。
|
|
324
|
+
// (REQUEST_DEFAULT_MAX_MESSAGES 只作为 resolveRequestBudget 里的**下限**存在,
|
|
325
|
+
// 不是本函数的默认值。)
|
|
326
|
+
export function prepareRequestMessages(messages, budget = {}) {
|
|
327
|
+
const { locale } = budget
|
|
328
|
+
const copy = buildRequestMessages(messages, budget)
|
|
329
|
+
stripStaleImages(copy, locale)
|
|
330
|
+
collapseTextParts(copy)
|
|
331
|
+
return sanitizeMessages(copy)
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// ── 上下文占用测量(给 UI 显示"这次请求占了多少")─────────────────────
|
|
335
|
+
//
|
|
336
|
+
// 为什么单独一个函数而不是让调用方自己算:
|
|
337
|
+
// · token 数没有真值 —— 只有 provider 返回的 usage 才是权威,而它在**响应回来之后**
|
|
338
|
+
// 才有。一轮工具循环里要在**请求发出去之前**就知道这次会带多少过去,只能估。
|
|
339
|
+
// · 估法必须一处。CLI(turn.js)与 Web(agentChat.js)各估一次必然漂移 —— 这仓库
|
|
340
|
+
// 已经被"同一口径写两遍"坑过(提示词四处、路径归一三处,失效方式是不报错、
|
|
341
|
+
// 两个入口表现不一样)。所以只有这一份。
|
|
342
|
+
//
|
|
343
|
+
// 这一节的估算函数已提到文件头部(裁剪算法要用,且 const 不会 hoist)。
|
|
344
|
+
// 为什么只留一份:CLI(turn.js)与 Web(agentChat.js)各估一次必然漂移 —— 这仓库
|
|
345
|
+
// 已经被"同一口径写两遍"坑过(提示词四处、路径归一三处,失效方式是不报错、
|
|
346
|
+
// 两个入口表现不一样)。精度只够画进度条/当闸门,不足以算钱;真要精确用量
|
|
347
|
+
// 请看 provider 报的 usage(CLI 的 /stats 有,Web 侧见 agentChat.js 的 context 事件)。
|
|
348
|
+
|
|
349
|
+
// 与 buildRequestMessages 里的 size() 同一口径:只算文本,不算图片 base64
|
|
350
|
+
// (图片是 base64,一张截图就上百万字符,计进预算会把进度条顶满)。
|
|
351
|
+
const messageTextSize = m => textOf(m).length + JSON.stringify(m.tool_calls || []).length
|
|
352
|
+
/** 消息的 token 估算:正文 + tool_calls 参数的序列化长度。图片不算。 */
|
|
353
|
+
const messageTokens = m => estimateTokens(textOf(m)) + estimateTokens(JSON.stringify(m.tool_calls || []))
|
|
354
|
+
|
|
355
|
+
/**
|
|
356
|
+
* 量一次请求的实际占用。给 UI 画圆环/进度条用,不参与任何裁剪决策。
|
|
357
|
+
*
|
|
358
|
+
* 主口径是 **token**(与裁剪算法同一个闸门),`chars` 只作附带信息。
|
|
359
|
+
* 之前主口径是字符,那是错的:用户看到"80,000 字符"根本不知道等于多少 token,
|
|
360
|
+
* 而模型窗口是按 token 计的。
|
|
361
|
+
*
|
|
362
|
+
* @param {Array} requestMessages 已经过 prepareRequestMessages 的**请求副本**
|
|
363
|
+
* @param {object} opts
|
|
364
|
+
* @param {number} opts.maxTokens 当前预算的 token 上限(画环的分母)
|
|
365
|
+
* @param {number} opts.maxMessages 当前预算的条数上限
|
|
366
|
+
* @param {number} [opts.maxChars] 当前预算的字符换算值(原样带回,供 UI 附带显示)
|
|
367
|
+
* @param {Array} [opts.transcript] 磁盘上的完整会话,算"被裁掉了多少"用
|
|
368
|
+
* @returns {{
|
|
369
|
+
* chars: number, tokens: number, estTokens: number, messages: number, images: number,
|
|
370
|
+
* maxTokens: number, maxMessages: number, maxChars: number | null,
|
|
371
|
+
* tokenRatio: number, messageRatio: number,
|
|
372
|
+
* transcriptMessages: number, transcriptChars: number, droppedMessages: number
|
|
373
|
+
* }}
|
|
374
|
+
*/
|
|
375
|
+
export function measureContextUsage(requestMessages, { maxTokens = REQUEST_DEFAULT_MAX_TOKENS, maxMessages = REQUEST_DEFAULT_MAX_MESSAGES, maxChars = null, transcript = null } = {}) {
|
|
376
|
+
const messages = Array.isArray(requestMessages) ? requestMessages : []
|
|
377
|
+
let chars = 0
|
|
378
|
+
let images = 0
|
|
379
|
+
let tokens = 0
|
|
380
|
+
for (const m of messages) {
|
|
381
|
+
chars += messageTextSize(m)
|
|
382
|
+
tokens += messageTokens(m)
|
|
383
|
+
if (Array.isArray(m?.content)) images += m.content.filter(p => p?.type === 'image_url').length
|
|
384
|
+
}
|
|
385
|
+
const transcriptChars = Array.isArray(transcript)
|
|
386
|
+
? transcript.reduce((n, m) => n + messageTextSize(m), 0)
|
|
387
|
+
: 0
|
|
388
|
+
return {
|
|
389
|
+
chars,
|
|
390
|
+
tokens,
|
|
391
|
+
messages: messages.length,
|
|
392
|
+
images,
|
|
393
|
+
estTokens: tokens,
|
|
394
|
+
maxTokens,
|
|
395
|
+
maxMessages,
|
|
396
|
+
// maxChars 原样带回(调用方传的是 resolveRequestBudget 的结果,里面有它)。
|
|
397
|
+
// 不重算 —— 调用方那份是按自己的预算解析出来的,重算会与实际裁剪用的不一致。
|
|
398
|
+
maxChars,
|
|
399
|
+
// 比率按 0~1 给,UI 拿它画进度/弧长就行,不必再除一遍
|
|
400
|
+
tokenRatio: maxTokens > 0 ? Math.min(tokens / maxTokens, 1) : 0,
|
|
401
|
+
messageRatio: maxMessages > 0 ? Math.min(messages.length / maxMessages, 1) : 0,
|
|
402
|
+
transcriptMessages: Array.isArray(transcript) ? transcript.length : messages.length,
|
|
403
|
+
transcriptChars,
|
|
404
|
+
droppedMessages: Array.isArray(transcript) ? Math.max(transcript.length - messages.length, 0) : 0,
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
// Only load explicitly named project instruction files, with bounded local redirects.
|
|
409
|
+
export async function loadProjectInstructions(cwd) {
|
|
410
|
+
const visited = new Set(), sections = []
|
|
411
|
+
let remaining = 16000
|
|
412
|
+
async function read(file, depth = 0) {
|
|
413
|
+
if (depth > 3 || remaining <= 0 || visited.has(file)) return
|
|
414
|
+
const relative = path.relative(cwd, file)
|
|
415
|
+
if (relative.startsWith('..') || path.isAbsolute(relative)) return
|
|
416
|
+
visited.add(file)
|
|
417
|
+
let raw
|
|
418
|
+
try {
|
|
419
|
+
const st = await fs.stat(file)
|
|
420
|
+
if (!st.isFile() || st.size > 128000) return
|
|
421
|
+
raw = await fs.readFile(file, 'utf8')
|
|
422
|
+
} catch { return }
|
|
423
|
+
const content = raw.slice(0, remaining)
|
|
424
|
+
remaining -= content.length
|
|
425
|
+
sections.push(`--- ${relative} ---\n${content}`)
|
|
426
|
+
for (const match of content.matchAll(/^@([^\r\n]+\.md)\s*$/gm)) await read(path.resolve(path.dirname(file), match[1].trim()), depth + 1)
|
|
427
|
+
}
|
|
428
|
+
await read(path.join(cwd, 'AGENTS.md'))
|
|
429
|
+
await read(path.join(cwd, 'CLAUDE.md'))
|
|
430
|
+
return sections.length ? '\n\n# Project instructions (apply within this project; user requests take precedence)\n' + sections.join('\n\n') : ''
|
|
431
|
+
}
|