mocode-ai 1.5.2 → 1.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/index.js +2 -0
- package/dist/agent/model-turn.js +42 -10
- package/dist/context/token-calibration.js +23 -2
- package/dist/i18n/index.js +4 -0
- package/dist/llm/index.js +74 -28
- package/dist/repl/status-bar.js +16 -2
- package/dist/session/compact.js +32 -12
- package/package.json +1 -1
package/dist/agent/index.js
CHANGED
|
@@ -382,6 +382,8 @@ runtimeOrContext = defaultRuntime) {
|
|
|
382
382
|
},
|
|
383
383
|
onStepStart: () => spinner.start(t('agent.thinking')),
|
|
384
384
|
onChatDone: () => spinner.stop(),
|
|
385
|
+
// 退避重试要在状态行可见:否则用户面对的是几十秒到几分钟的静止 spinner,与卡死无异。
|
|
386
|
+
onModelRetry: (r) => spinner.start(t('agent.retrying', { seconds: Math.max(1, Math.round(r.waitMs / 1000)), attempt: r.attempt })),
|
|
385
387
|
// 流式实时用量 → 底栏 context 进度条左侧 chip;轮末由 repl 清空。
|
|
386
388
|
onLiveUsage: (u) => layout.setLiveUsage(u),
|
|
387
389
|
onTextEnd: () => {
|
package/dist/agent/model-turn.js
CHANGED
|
@@ -1,4 +1,20 @@
|
|
|
1
1
|
import { estimatePromptTokens, estimateTokens, isContextLengthError, } from '../llm/index.js';
|
|
2
|
+
/**
|
|
3
|
+
* 沿 cause 链(≤3 层)取第一个 errno,供 trace 取证。
|
|
4
|
+
* undici 把底层 errno 挂在 cause 上(`TypeError: fetch failed` → cause `read ECONNRESET`),
|
|
5
|
+
* openai SDK 再包一层 APIConnectionError 后,顶层 code 恒为 undefined;不留这一项,
|
|
6
|
+
* 事后只能看到一个 'Error',无法区分 DNS 失败 / 连接被拒 / TLS 握手失败。
|
|
7
|
+
*/
|
|
8
|
+
function traceCauseCode(err) {
|
|
9
|
+
let cur = err;
|
|
10
|
+
for (let depth = 0; depth <= 3 && cur && typeof cur === 'object'; depth++) {
|
|
11
|
+
const e = cur;
|
|
12
|
+
if (typeof e.code === 'string')
|
|
13
|
+
return e.code;
|
|
14
|
+
cur = e.cause;
|
|
15
|
+
}
|
|
16
|
+
return undefined;
|
|
17
|
+
}
|
|
2
18
|
/** Executes context preparation plus exactly one model step, including the single overflow retry path. */
|
|
3
19
|
export async function runModelTurn(input) {
|
|
4
20
|
const { opts, ctx, history, historyManager, runtimeContextState, scheduler, contextTrimmer, modelRunner, activeTools, runPolicy, step, cacheState, turnLifecycle, cancellationLifecycle, rebuildHistoryIndexes, } = input;
|
|
@@ -112,28 +128,44 @@ export async function runModelTurn(input) {
|
|
|
112
128
|
onText,
|
|
113
129
|
onToolCall,
|
|
114
130
|
onProgress: reportLive,
|
|
115
|
-
onRetry: (retry) =>
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
131
|
+
onRetry: (retry) => {
|
|
132
|
+
emitTrace('model_retry', {
|
|
133
|
+
model: requestModel,
|
|
134
|
+
provider,
|
|
135
|
+
attempt: retry.attempt,
|
|
136
|
+
nextAttempt: retry.nextAttempt,
|
|
137
|
+
waitMs: retry.waitMs,
|
|
138
|
+
code: retry.code,
|
|
139
|
+
});
|
|
140
|
+
// 退避最长可到 30s、最坏累计数分钟。只写 trace 的话用户看到的就是「卡住不动」,
|
|
141
|
+
// 与「直接报错终止」一样不可解释 —— 所以同时转达给宿主做可见反馈。
|
|
142
|
+
hooks.onModelRetry?.(retry);
|
|
143
|
+
},
|
|
123
144
|
};
|
|
124
145
|
const runChatOnce = async () => {
|
|
125
146
|
try {
|
|
126
147
|
return await modelRunner.run({ history: requestHistory, handlers: chatHandlers, tools: activeTools }, signal);
|
|
127
148
|
}
|
|
128
149
|
catch (error) {
|
|
129
|
-
const errorValue = error && typeof error === 'object'
|
|
150
|
+
const errorValue = error && typeof error === 'object'
|
|
151
|
+
? error
|
|
152
|
+
: undefined;
|
|
130
153
|
emitTrace('model_end', {
|
|
131
154
|
model: requestModel,
|
|
132
155
|
provider,
|
|
133
156
|
status: signal?.aborted ? 'aborted' : 'error',
|
|
134
157
|
code: typeof errorValue?.status === 'number'
|
|
135
158
|
? `HTTP_${errorValue.status}`
|
|
136
|
-
:
|
|
159
|
+
: // 构造器名优先于 `name`:SDK 的 APIError 家族 err.name 恒为 'Error'(无信息量),
|
|
160
|
+
// 构造器名才分得出 APIConnectionError(建连失败)/ APIError(流内报错)。
|
|
161
|
+
(errorValue?.code ??
|
|
162
|
+
error?.constructor?.name ??
|
|
163
|
+
errorValue?.name ??
|
|
164
|
+
'MODEL_ERROR'),
|
|
165
|
+
// 取证用:错误文案与 cause 链上的 errno。没有这两项时,trace 只留一个 'Error',
|
|
166
|
+
// 事后无法判断到底是 DNS 失败、连接被拒还是 TLS 握手失败(实测踩过)。
|
|
167
|
+
message: typeof errorValue?.message === 'string' ? errorValue.message.slice(0, 300) : undefined,
|
|
168
|
+
causeCode: traceCauseCode(error),
|
|
137
169
|
durationMs: Date.now() - modelStartedAt,
|
|
138
170
|
});
|
|
139
171
|
throw error;
|
|
@@ -2,11 +2,27 @@ import fs from 'node:fs';
|
|
|
2
2
|
import os from 'node:os';
|
|
3
3
|
import path from 'node:path';
|
|
4
4
|
import { createHash } from 'node:crypto';
|
|
5
|
-
|
|
5
|
+
// v2:引入 SUSPECT_* 可信区间。旧缓存里被口径不可比的 provider 砸到 MIN_CORRECTION
|
|
6
|
+
// 下限、又被 clamp 住的条目(correction=0.5 且再不会有新样本去修正它)会永久打对折
|
|
7
|
+
// 所有显示数字,只能整表作废——丢掉的是几十个样本,重学只要几步。
|
|
8
|
+
const CACHE_VERSION = 2;
|
|
6
9
|
const EWMA_ALPHA = 0.2;
|
|
7
10
|
const MIN_CORRECTION = 0.5;
|
|
8
11
|
const MAX_CORRECTION = 2;
|
|
9
12
|
const MAX_ENTRIES = 64;
|
|
13
|
+
/**
|
|
14
|
+
* 可信样本区间(actual / estimated)。超出即**不并入 EWMA**。
|
|
15
|
+
*
|
|
16
|
+
* 为什么需要这道闸:估算器的任务只是「别让请求溢出窗口」,它允许偏高;而 usage 是
|
|
17
|
+
* provider 报的账,两者本该同量级(本机 40+ 会话实测 est/actual 落在 0.33–1.66)。
|
|
18
|
+
* 一旦某个 gateway/model 的 usage 口径不可比(实测踩过:localhost 网关的 thinking 模型
|
|
19
|
+
* 报 20.8 chars/token,同机其它 provider 全是 1.3–3.3),ratio 会直接砸到 MIN_CORRECTION
|
|
20
|
+
* 下限并被 clamp 住——之后 correction 恒为 0.5,**每个乘以它的显示数字都被无谓地打对折**
|
|
21
|
+
* (压缩行 40% vs 底栏 80%,用户看到的两个数都不是真值)。这种样本学不出有用信息,
|
|
22
|
+
* 只会污染 UI;丢掉它,correction 保持上一次可用值(或 1)。
|
|
23
|
+
*/
|
|
24
|
+
const SUSPECT_MIN_RATIO = 0.3;
|
|
25
|
+
const SUSPECT_MAX_RATIO = 3.5;
|
|
10
26
|
let cache;
|
|
11
27
|
const toolFingerprints = new WeakMap();
|
|
12
28
|
function cachePath() {
|
|
@@ -75,7 +91,8 @@ export function getTokenCalibration(baseURL, model, tools) {
|
|
|
75
91
|
const entry = readCache().entries[calibrationKey(baseURL, model, tools)];
|
|
76
92
|
return validEntry(entry) ? { correction: entry.correction, samples: entry.samples } : { correction: 1, samples: 0 };
|
|
77
93
|
}
|
|
78
|
-
/** 用一次真实 prompt usage 更新 EWMA;只落比例和样本数,不保存任何消息内容。
|
|
94
|
+
/** 用一次真实 prompt usage 更新 EWMA;只落比例和样本数,不保存任何消息内容。
|
|
95
|
+
* 样本与估算器差到 SUSPECT_* 区间之外时判为 provider 口径异常,直接丢弃(不改 correction)。 */
|
|
79
96
|
export function updateTokenCalibration(baseURL, model, tools, estimatedTokens, actualTokens) {
|
|
80
97
|
if (estimatedTokens <= 100 ||
|
|
81
98
|
actualTokens <= 100 ||
|
|
@@ -83,6 +100,10 @@ export function updateTokenCalibration(baseURL, model, tools, estimatedTokens, a
|
|
|
83
100
|
!Number.isFinite(actualTokens)) {
|
|
84
101
|
return getTokenCalibration(baseURL, model, tools);
|
|
85
102
|
}
|
|
103
|
+
const ratio = actualTokens / estimatedTokens;
|
|
104
|
+
if (ratio < SUSPECT_MIN_RATIO || ratio > SUSPECT_MAX_RATIO) {
|
|
105
|
+
return getTokenCalibration(baseURL, model, tools);
|
|
106
|
+
}
|
|
86
107
|
const key = calibrationKey(baseURL, model, tools);
|
|
87
108
|
const store = readCache();
|
|
88
109
|
const previous = store.entries[key];
|
package/dist/i18n/index.js
CHANGED
|
@@ -186,10 +186,12 @@ const zhCN = {
|
|
|
186
186
|
'status.measured': '实测',
|
|
187
187
|
'status.estimated': '估算',
|
|
188
188
|
'status.messages': '{count} 条消息',
|
|
189
|
+
'status.providerMeasured': 'provider 上一步实测 {tokens}:{pct}%',
|
|
189
190
|
'agent.sending': '发送中… (任意键 / Esc / Ctrl+C 撤回)',
|
|
190
191
|
'agent.thinking': '思考中',
|
|
191
192
|
'agent.generating': '生成 {tool}',
|
|
192
193
|
'agent.executing': '执行 {tool}',
|
|
194
|
+
'agent.retrying': '连接异常,{seconds}s 后重试(第 {attempt} 次)',
|
|
193
195
|
'agent.noReply': '(无回复)',
|
|
194
196
|
'agent.maxSteps': '达到最大步数({count}),本轮停止。',
|
|
195
197
|
'agent.aborted': '(已中断)',
|
|
@@ -515,10 +517,12 @@ const en = {
|
|
|
515
517
|
'status.measured': 'measured',
|
|
516
518
|
'status.estimated': 'estimated',
|
|
517
519
|
'status.messages': '{count} messages',
|
|
520
|
+
'status.providerMeasured': 'provider measured {tokens} last step:{pct}%',
|
|
518
521
|
'agent.sending': 'Sending… (any key / Esc / Ctrl+C to recall)',
|
|
519
522
|
'agent.thinking': 'Thinking',
|
|
520
523
|
'agent.generating': 'Generating {tool}',
|
|
521
524
|
'agent.executing': 'Running {tool}',
|
|
525
|
+
'agent.retrying': 'Connection lost, retrying in {seconds}s (attempt {attempt})',
|
|
522
526
|
'agent.noReply': '(no reply)',
|
|
523
527
|
'agent.maxSteps': 'Maximum steps reached ({count}); this turn has stopped.',
|
|
524
528
|
'agent.aborted': '(aborted)',
|
package/dist/llm/index.js
CHANGED
|
@@ -107,7 +107,9 @@ export function isRetryableError(err, signal) {
|
|
|
107
107
|
if (!err || typeof err !== 'object')
|
|
108
108
|
return false;
|
|
109
109
|
const e = err;
|
|
110
|
-
|
|
110
|
+
// SDK 的 APIError 家族不设 this.name(见下),判类别一律用构造器名。
|
|
111
|
+
const ctor = e.constructor?.name ?? '';
|
|
112
|
+
if (e.name === 'AbortError' || e.name === 'APIUserAbortError' || ctor === 'APIUserAbortError')
|
|
111
113
|
return false;
|
|
112
114
|
// OpenAI SDK APIError 走 status 分支(覆盖 4xx/5xx/429)
|
|
113
115
|
const status = e.status;
|
|
@@ -118,23 +120,26 @@ export function isRetryableError(err, signal) {
|
|
|
118
120
|
return true;
|
|
119
121
|
return false;
|
|
120
122
|
}
|
|
121
|
-
//
|
|
122
|
-
|
|
123
|
-
if (code
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
code === 'ECONNREFUSED' ||
|
|
128
|
-
code === 'EPIPE') {
|
|
123
|
+
// 证书 / 协议层不匹配是**永久性**错误,重试只会白等满退避(10 次≈两分钟)。
|
|
124
|
+
// 必须先于下面的宽兜底判定:fetch failed / cause 里的证书错文案都会被宽兜底捞走。
|
|
125
|
+
if (causeChainFrames(err).some((f) => (!!f.code && FATAL_TRANSPORT_CODE.test(f.code)) || (!!f.message && FATAL_TRANSPORT_MESSAGE.test(f.message))))
|
|
126
|
+
return false;
|
|
127
|
+
// Node 网络错 errno:顶层没有就沿 cause 链找(undici 把 errno 藏在 cause 里)。
|
|
128
|
+
if (causeChainFrames(err).some((f) => !!f.code && RETRYABLE_ERRNO.has(f.code)))
|
|
129
129
|
return true;
|
|
130
|
-
|
|
131
|
-
//
|
|
132
|
-
|
|
130
|
+
// OpenAI SDK 的网络错类(无 status)。必须用**构造器名**:SDK 的 APIError 家族只做
|
|
131
|
+
// `super(message)`,从不设 this.name(`err.name` 恒为 'Error')—— 旧代码这里写
|
|
132
|
+
// `e.name === 'APIConnectionError'` 是永不命中的死分支,于是建连失败(DNS / 连接被拒 /
|
|
133
|
+
// TLS 握手 / 半路断流)一次即抛、整轮 run 直接终止(trace 里 model_end.code 只剩 'Error')。
|
|
134
|
+
if (ctor === 'APIConnectionError' || ctor === 'APIConnectionTimeoutError')
|
|
133
135
|
return true;
|
|
134
|
-
//
|
|
135
|
-
|
|
136
|
+
// 兜底:文案。部分代理把错误折叠成普通 Error;SDK 的 APIConnectionError 默认文案就是
|
|
137
|
+
// 'Connection error.'(负载里没有任何细节,`code` 也为 undefined,只能靠文案兜)。
|
|
138
|
+
const msg = typeof e.message === 'string' ? e.message : '';
|
|
139
|
+
if (/\btime(d|ed)?\s*out\b|ETIMEDOUT/i.test(msg))
|
|
140
|
+
return true;
|
|
141
|
+
if (/^connection error\.?$/i.test(msg.trim()) || /\bfetch failed\b/i.test(msg))
|
|
136
142
|
return true;
|
|
137
|
-
}
|
|
138
143
|
return false;
|
|
139
144
|
}
|
|
140
145
|
/**
|
|
@@ -161,16 +166,54 @@ const STREAM_BREAK_CODES = new Set([
|
|
|
161
166
|
'ERR_SOCKET_CONNECTION_TIMEOUT',
|
|
162
167
|
]);
|
|
163
168
|
const STREAM_BREAK_MESSAGE = /premature close|other side closed|socket hang up|\bterminated\b|stream (?:closed|ended) (?:prematurely|unexpectedly)|econnreset|econnaborted/i;
|
|
164
|
-
/**
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
169
|
+
/**
|
|
170
|
+
* 「连接层」可重试 errno。与 STREAM_BREAK_CODES 的区别:这些在**建连/发请求**阶段就失败
|
|
171
|
+
* (DNS 解析、拒绝连接、路由不可达、连接超时),压根没有响应流可言,但处置一样 —— 重试。
|
|
172
|
+
* 旧实现只在顶层 `err.code` 上查,而 undici 把 errno 埋在 cause 里,永远查不到。
|
|
173
|
+
*/
|
|
174
|
+
const RETRYABLE_ERRNO = new Set([
|
|
175
|
+
'ETIMEDOUT',
|
|
176
|
+
'ECONNRESET',
|
|
177
|
+
'ENOTFOUND',
|
|
178
|
+
'EAI_AGAIN',
|
|
179
|
+
'ECONNREFUSED',
|
|
180
|
+
'EPIPE',
|
|
181
|
+
'ECONNABORTED',
|
|
182
|
+
'ENETUNREACH',
|
|
183
|
+
'EHOSTUNREACH',
|
|
184
|
+
'ERR_SOCKET_CONNECTION_TIMEOUT',
|
|
185
|
+
'UND_ERR_CONNECT_TIMEOUT',
|
|
186
|
+
'UND_ERR_SOCKET',
|
|
187
|
+
]);
|
|
188
|
+
/** 证书 / TLS 协议不匹配:重试必然再错,判死以免白等退避。 */
|
|
189
|
+
const FATAL_TRANSPORT_CODE = /^(?:ERR_SSL|ERR_TLS|ERR_OSSL|UNABLE_TO_VERIFY|DEPTH_ZERO|SELF_SIGNED|CERT_|EPROTO)/;
|
|
190
|
+
const FATAL_TRANSPORT_MESSAGE = /certificate|self[- ]signed|\bEPROTO\b|wrong version number|unsupported protocol/i;
|
|
191
|
+
/**
|
|
192
|
+
* 沿错误自身与 cause 链(≤3 层)收集每层的 code / message。
|
|
193
|
+
*
|
|
194
|
+
* 为什么必须看 cause:undici 把底层 errno 挂在 cause 上(`TypeError: fetch failed` →
|
|
195
|
+
* cause `Error: read ECONNRESET`,code 在 cause 里),openai SDK 又把那次 fetch 失败再包一层
|
|
196
|
+
* `APIConnectionError({ cause })`(`core.mjs` 的 catch 分支)。只看顶层 `err.code` / `err.message`
|
|
197
|
+
* 永远是 undefined / 'Connection error.' —— 这正是这类故障长期无法被识别、无法重试的根因。
|
|
198
|
+
*/
|
|
199
|
+
function causeChainFrames(err) {
|
|
200
|
+
const frames = [];
|
|
201
|
+
let cur = err;
|
|
202
|
+
for (let depth = 0; depth <= 3; depth++) {
|
|
203
|
+
if (!cur || typeof cur !== 'object')
|
|
204
|
+
break;
|
|
205
|
+
const e = cur;
|
|
206
|
+
frames.push({
|
|
207
|
+
code: typeof e.code === 'string' ? e.code : undefined,
|
|
208
|
+
message: typeof e.message === 'string' ? e.message : undefined,
|
|
209
|
+
});
|
|
210
|
+
cur = e.cause;
|
|
211
|
+
}
|
|
212
|
+
return frames;
|
|
213
|
+
}
|
|
214
|
+
/** 错误链(含 cause)上是否存在「连接被掐断」的信号。 */
|
|
215
|
+
function hasStreamBreakSignal(err) {
|
|
216
|
+
return causeChainFrames(err).some((f) => (!!f.code && STREAM_BREAK_CODES.has(f.code)) || (!!f.message && STREAM_BREAK_MESSAGE.test(f.message)));
|
|
174
217
|
}
|
|
175
218
|
/**
|
|
176
219
|
* 判定「服务端在响应流中途报错 / 流被中途掐断」——HTTP 层已建连并开始流式返回(状态码 200),
|
|
@@ -186,8 +229,9 @@ function hasStreamBreakSignal(err, depth = 0) {
|
|
|
186
229
|
* 这三类都不沾 → 被当成「不可重试的客户端请求错」一次即抛,整轮 run 直接终止。但它们的真实语义
|
|
187
230
|
* 是「服务端/链路在生成到一半时挂了」,属瞬时故障,重试是正确处置(已实测同一会话更大 prompt 可成功)。
|
|
188
231
|
*
|
|
189
|
-
* 只认 APIError 本身,不认子类:APIConnectionError / APIConnectionTimeoutError
|
|
190
|
-
* isRetryableError
|
|
232
|
+
* 只认 APIError 本身,不认子类:APIConnectionError / APIConnectionTimeoutError 是**建连阶段**
|
|
233
|
+
* 失败,归 isRetryableError 的构造器名分支(它有 10 次 HTTP 层预算,更合适);而 APIUserAbortError
|
|
234
|
+
* 是用户中断,绝不能重试。
|
|
191
235
|
* 判据用构造器名而非 instanceof:同进程若存在 openai 的多份模块实例(ESM/CJS 混载),
|
|
192
236
|
* instanceof 会失配。
|
|
193
237
|
*
|
|
@@ -424,7 +468,9 @@ function retryErrorCode(error) {
|
|
|
424
468
|
const value = error;
|
|
425
469
|
if (typeof value.status === 'number')
|
|
426
470
|
return `HTTP_${value.status}`;
|
|
427
|
-
|
|
471
|
+
// 优先构造器名:SDK 的 APIError 家族 `err.name` 恒为 'Error'(无信息量),
|
|
472
|
+
// 构造器名才区分得出是 APIConnectionError(建连失败)还是流内 APIError。
|
|
473
|
+
return value.code ?? value.constructor?.name ?? value.name ?? 'RETRYABLE_ERROR';
|
|
428
474
|
}
|
|
429
475
|
/** Bind chat dispatch, retries and built-in providers to one explicit runtime. */
|
|
430
476
|
export function createChatTransport(runtime) {
|
package/dist/repl/status-bar.js
CHANGED
|
@@ -23,8 +23,20 @@ export function renderContextBar(history) {
|
|
|
23
23
|
const W = 10;
|
|
24
24
|
const filled = Math.round(pct * W);
|
|
25
25
|
const bar = '█'.repeat(filled) + '░'.repeat(W - filled);
|
|
26
|
-
const src = contextState.lastUsage ? t('status.measured') : t('status.estimated');
|
|
27
26
|
const k = (n) => `${Math.round(n / 1000)}k`;
|
|
27
|
+
// 标签必须描述**这个数字**是什么:本条 bar 恒为「对话内容」估算(dialog-only,不含
|
|
28
|
+
// system prompt / 工具 schema),与底栏用量条、80% 压力线都不是同一个数。
|
|
29
|
+
// 旧实现只要 lastUsage 存在就打「实测」,而显示的仍是估算值 —— 标签与数字对不上,
|
|
30
|
+
// 用户拿它跟底栏对账只会更困惑。改为:数字照旧标「估算」,把 provider 真正测到的
|
|
31
|
+
// prompt 并列出来,两者差多少一眼可见(差值本身就是 provider 口径是否可信的证据)。
|
|
32
|
+
const src = t('status.estimated');
|
|
33
|
+
const measured = contextState.lastUsage?.promptTokens;
|
|
34
|
+
const measuredNote = measured && measured > 0
|
|
35
|
+
? ` · ${t('status.providerMeasured', {
|
|
36
|
+
tokens: k(measured),
|
|
37
|
+
pct: Math.round(Math.min(1, measured / win) * 100),
|
|
38
|
+
})}`
|
|
39
|
+
: '';
|
|
28
40
|
const pctCol = pct >= DEFAULT_BUDGET_POLICY.pressureTriggerRatio ? ui.yellow : ui.accent;
|
|
29
41
|
const lifecycle = contextState.lifecycleStats;
|
|
30
42
|
const archived = computePruneStats(history);
|
|
@@ -36,7 +48,7 @@ export function renderContextBar(history) {
|
|
|
36
48
|
? `\n lifecycle · live ${lifecycle.live} · referenced ${lifecycle.referenced} · digested ${lifecycle.digested} · stubbed ${lifecycle.stubbed}`
|
|
37
49
|
: '\n lifecycle · no active snapshot (run a tool-enabled turn first)';
|
|
38
50
|
const archiveLine = `\n archived tool results · ${archived.stubbed}`;
|
|
39
|
-
return `${ui.gray}[${pctCol}${bar}${ui.reset}] ${Math.round(pct * 100)}% ${k(est)}/${k(win)} tokens · ${t('status.messages', { count: history.length })} (${src})${ui.reset}${artifactLine}${lifecycleLine}${archiveLine}`;
|
|
51
|
+
return `${ui.gray}[${pctCol}${bar}${ui.reset}] ${Math.round(pct * 100)}% ${k(est)}/${k(win)} tokens · ${t('status.messages', { count: history.length })} (${src})${measuredNote}${ui.reset}${artifactLine}${lifecycleLine}${archiveLine}`;
|
|
40
52
|
}
|
|
41
53
|
/** 状态行用量条(精简版,进底栏):[bar] pct% k/k。
|
|
42
54
|
* 必须**用全 prompt 估算**(消息 + 工具 schema + 尾部 ephemeral 注入),与压缩触发器
|
|
@@ -45,6 +57,8 @@ export function renderContextBar(history) {
|
|
|
45
57
|
* 触发器用 `Math.max(rawTotal, total) >= 0.8 * window`,bar 也照搬:校正后和校正前
|
|
46
58
|
* 哪个大取哪个,确保不会因 correction<1 而低估。ephemeral 文本由 agent/core 每步写入
|
|
47
59
|
* contextState.ephemeralText(避免在 bar 里再读一次 notes.md)。
|
|
60
|
+
* 压缩行(compact.ts 的 compactionLogLine)已统一到同一口径:数字取裸估算、后缀带窗口
|
|
61
|
+
* 百分比,两者可直接对账——「底栏 80% 而压缩行 40%」这类口径分裂不再出现。
|
|
48
62
|
*
|
|
49
63
|
* /context 命令仍是 dialog-only(见 renderContextBar):它的设计意图是"我说了多少"而非
|
|
50
64
|
* "还剩多少空间",两条职责分开。 */
|
package/dist/session/compact.js
CHANGED
|
@@ -559,6 +559,26 @@ async function defaultSummarize(older, focus, signal, runtime = defaultCompactio
|
|
|
559
559
|
}
|
|
560
560
|
}
|
|
561
561
|
// ── 对外:compactHistory / maybeCompact ───────────────────────────────────
|
|
562
|
+
/**
|
|
563
|
+
* 用户可见的上下文占用口径:**裸估算**(不乘 correction)。
|
|
564
|
+
*
|
|
565
|
+
* 为什么必须裸估算:触发判定用的就是它——session/scheduler.ts 的 80% 压力线比
|
|
566
|
+
* `Math.max(report.rawTotal, report.total)`(context/budget.ts scheduleActions),
|
|
567
|
+
* repl 底栏用量条也取 `max(raw, corrected)`(repl/status-bar.ts)。压缩行若打印校正后的值,
|
|
568
|
+
* 同一时刻 TUI 上就同时存在三个都叫 token 的数字(provider 实测 / 校正后 / 触发用裸估),
|
|
569
|
+
* 用户无从判断哪条线会触发 —— 实测踩过:底栏 80%、压缩行 40%,看起来「没到线却压了」。
|
|
570
|
+
* correction 只服务内部保留区尺寸估算(见 correctTokenEstimate 调用点),不再外泄成占用读数。
|
|
571
|
+
*/
|
|
572
|
+
function rawPromptTokens(history, activeTools) {
|
|
573
|
+
return estimatePromptTokens(history, activeTools, 1);
|
|
574
|
+
}
|
|
575
|
+
/** 压缩行统一格式:`● 标签 205214 → 28614 tokens (80% → 11%)`。
|
|
576
|
+
* 百分比与数字同源(裸估算 / 窗口),让「凭什么这时压」在行内自证,不必回头翻代码。 */
|
|
577
|
+
function compactionLogLine(label, before, after, window) {
|
|
578
|
+
const pct = (n) => `${Math.round((n / Math.max(1, window)) * 100)}%`;
|
|
579
|
+
return (` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}${label}${ui.reset} ` +
|
|
580
|
+
`${ui.dim}${before} → ${after} tokens (${pct(before)} → ${pct(after)})${ui.reset}\n`);
|
|
581
|
+
}
|
|
562
582
|
/**
|
|
563
583
|
* 压缩 history(原地)。手动 /compact 与自动 maybeCompact 都走这里。
|
|
564
584
|
* 不检查阈值——调用方(maybeCompact)决定是否调;/compact 直接调以强制压缩。
|
|
@@ -566,8 +586,8 @@ async function defaultSummarize(older, focus, signal, runtime = defaultCompactio
|
|
|
566
586
|
export async function compactHistory(history, opts) {
|
|
567
587
|
const state = opts.contextState ?? contextState;
|
|
568
588
|
const activeTools = opts.tools ?? chatTools;
|
|
569
|
-
const estimateBefore =
|
|
570
|
-
state.lastEstimate =
|
|
589
|
+
const estimateBefore = rawPromptTokens(history, activeTools);
|
|
590
|
+
state.lastEstimate = estimatePromptTokens(history, activeTools, state.correction);
|
|
571
591
|
// 调用前就已中断(用户在上一步末尾按的 Ctrl+C):一步都别做,直接冒泡。
|
|
572
592
|
// 不做完再抛是为了保证 history 完全未被触碰——abortRestore 才还原得干净。
|
|
573
593
|
if (opts.signal?.aborted) {
|
|
@@ -688,12 +708,12 @@ export async function compactHistory(history, opts) {
|
|
|
688
708
|
const single = groups[0];
|
|
689
709
|
const userOnly = single.tools.length === 0 && single.assistant?.role === 'user';
|
|
690
710
|
if (!userOnly && microcompactGroup(single)) {
|
|
691
|
-
const estimateAfter =
|
|
692
|
-
state.lastEstimate =
|
|
711
|
+
const estimateAfter = rawPromptTokens(history, activeTools);
|
|
712
|
+
state.lastEstimate = estimatePromptTokens(history, activeTools, state.correction);
|
|
693
713
|
state.lastUsage = undefined;
|
|
694
714
|
if (!layout.isLastContentRowBlank())
|
|
695
715
|
layout.contentWrite('\n');
|
|
696
|
-
layout.contentWrite(
|
|
716
|
+
layout.contentWrite(compactionLogLine('强制微压缩(单组)', estimateBefore, estimateAfter, opts.window));
|
|
697
717
|
return {
|
|
698
718
|
compacted: true,
|
|
699
719
|
summarized: false,
|
|
@@ -780,14 +800,14 @@ export async function compactHistory(history, opts) {
|
|
|
780
800
|
history.length = 0;
|
|
781
801
|
history.push(...rebuilt);
|
|
782
802
|
(opts.runtime?.rollbackStore ?? defaultRollbackStore).pruneAfterCompaction(history);
|
|
783
|
-
const estimateAfter =
|
|
784
|
-
state.lastEstimate =
|
|
785
|
-
state.lastUsage = undefined; // 压缩后旧 usage 失效,/context
|
|
803
|
+
const estimateAfter = rawPromptTokens(history, activeTools);
|
|
804
|
+
state.lastEstimate = estimatePromptTokens(history, activeTools, state.correction);
|
|
805
|
+
state.lastUsage = undefined; // 压缩后旧 usage 失效,/context 改用估算
|
|
786
806
|
// 压缩行与上一个工具批次摘要行之间补空行分隔(compact 在 core step 循环顶部触发,
|
|
787
807
|
// 上一步的 batch 可能尚未 flush,缓冲末行仍是 ● 工具摘要行 → 两行黏在一起)。
|
|
788
808
|
if (!layout.isLastContentRowBlank())
|
|
789
809
|
layout.contentWrite('\n');
|
|
790
|
-
layout.contentWrite(
|
|
810
|
+
layout.contentWrite(compactionLogLine('压缩上下文', estimateBefore, estimateAfter, opts.window));
|
|
791
811
|
// 抖动保护:压缩后仍超阈 → 提示 /clear,不死循环
|
|
792
812
|
if (estimateAfter >= opts.threshold * opts.window) {
|
|
793
813
|
layout.contentWrite(` ${ui.yellow}●${ui.reset} ${ui.yellow}压缩后仍超阈,可能存在超大单条;建议 /clear。${ui.reset}\n`);
|
|
@@ -811,13 +831,13 @@ export async function compactHistory(history, opts) {
|
|
|
811
831
|
if (microcompactGroup(g))
|
|
812
832
|
microcompactDone = true;
|
|
813
833
|
}
|
|
814
|
-
const estimateAfter =
|
|
815
|
-
state.lastEstimate =
|
|
834
|
+
const estimateAfter = rawPromptTokens(history, activeTools);
|
|
835
|
+
state.lastEstimate = estimatePromptTokens(history, activeTools, state.correction);
|
|
816
836
|
state.lastUsage = undefined; // token 数已变,旧 usage 失效
|
|
817
837
|
if (microcompactDone) {
|
|
818
838
|
if (!layout.isLastContentRowBlank())
|
|
819
839
|
layout.contentWrite('\n');
|
|
820
|
-
layout.contentWrite(
|
|
840
|
+
layout.contentWrite(compactionLogLine('微压缩旧工具结果', estimateBefore, estimateAfter, opts.window));
|
|
821
841
|
return {
|
|
822
842
|
compacted: true,
|
|
823
843
|
summarized: false,
|