mocode-ai 0.6.5 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/core.js +54 -31
- package/dist/context/age-aware.js +136 -0
- package/dist/context/budget.js +56 -86
- package/dist/context/encoders/code.js +186 -60
- package/dist/context/encoders/command.js +201 -0
- package/dist/context/encoders/graph.js +69 -19
- package/dist/context/encoders/index.js +2 -2
- package/dist/context/encoders/log.js +2 -52
- package/dist/context/encoders/search.js +116 -50
- package/dist/context/index.js +1 -1
- package/dist/context/pipeline.js +5 -1
- package/dist/context/relevance.js +166 -121
- package/dist/context/token-calibration.js +104 -0
- package/dist/llm/index.js +42 -99
- package/dist/llm/think-filter.js +90 -0
- package/dist/repl/index.js +0 -2
- package/dist/session/compact.js +20 -23
- package/dist/session/index.js +3 -3
- package/dist/session/scheduler.js +18 -38
- package/dist/tools/builtins/run-command.js +29 -9
- package/dist/tools/constants.js +0 -17
- package/package.json +1 -1
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
import fs from 'node:fs';
|
|
2
|
+
import os from 'node:os';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import { createHash } from 'node:crypto';
|
|
5
|
+
const CACHE_VERSION = 1;
|
|
6
|
+
const EWMA_ALPHA = 0.2;
|
|
7
|
+
const MIN_CORRECTION = 0.5;
|
|
8
|
+
const MAX_CORRECTION = 2;
|
|
9
|
+
const MAX_ENTRIES = 64;
|
|
10
|
+
let cache;
|
|
11
|
+
const toolFingerprints = new WeakMap();
|
|
12
|
+
function cachePath() {
|
|
13
|
+
return process.env.MOCODE_TOKEN_CALIBRATION_CACHE
|
|
14
|
+
|| path.join(os.homedir(), '.mocode', 'token-calibration.json');
|
|
15
|
+
}
|
|
16
|
+
function hash(value) {
|
|
17
|
+
return createHash('sha256').update(value).digest('hex');
|
|
18
|
+
}
|
|
19
|
+
function toolFingerprint(tools) {
|
|
20
|
+
const objectKey = tools;
|
|
21
|
+
const hit = toolFingerprints.get(objectKey);
|
|
22
|
+
if (hit)
|
|
23
|
+
return hit;
|
|
24
|
+
const fingerprint = hash(JSON.stringify(tools));
|
|
25
|
+
toolFingerprints.set(objectKey, fingerprint);
|
|
26
|
+
return fingerprint;
|
|
27
|
+
}
|
|
28
|
+
function calibrationKey(baseURL, model, tools) {
|
|
29
|
+
// 只把摘要写盘,避免 URL 中偶然携带的凭据出现在缓存文件。
|
|
30
|
+
return hash(`${baseURL}\0${model}\0${toolFingerprint(tools)}`);
|
|
31
|
+
}
|
|
32
|
+
function readCache() {
|
|
33
|
+
if (cache)
|
|
34
|
+
return cache;
|
|
35
|
+
try {
|
|
36
|
+
const parsed = JSON.parse(fs.readFileSync(cachePath(), 'utf8'));
|
|
37
|
+
if (parsed.version === CACHE_VERSION && parsed.entries && typeof parsed.entries === 'object') {
|
|
38
|
+
cache = parsed;
|
|
39
|
+
return cache;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
// 不存在或损坏都从空缓存开始;校准是增强能力,不能阻断请求。
|
|
44
|
+
}
|
|
45
|
+
cache = { version: CACHE_VERSION, entries: {} };
|
|
46
|
+
return cache;
|
|
47
|
+
}
|
|
48
|
+
function writeCache() {
|
|
49
|
+
if (!cache)
|
|
50
|
+
return;
|
|
51
|
+
try {
|
|
52
|
+
const entries = Object.entries(cache.entries)
|
|
53
|
+
.sort(([, a], [, b]) => b.updatedAt - a.updatedAt)
|
|
54
|
+
.slice(0, MAX_ENTRIES);
|
|
55
|
+
cache.entries = Object.fromEntries(entries);
|
|
56
|
+
const target = cachePath();
|
|
57
|
+
fs.mkdirSync(path.dirname(target), { recursive: true });
|
|
58
|
+
const tmp = `${target}.tmp-${process.pid}`;
|
|
59
|
+
fs.writeFileSync(tmp, JSON.stringify(cache), 'utf8');
|
|
60
|
+
fs.renameSync(tmp, target);
|
|
61
|
+
}
|
|
62
|
+
catch {
|
|
63
|
+
// 只影响跨进程复用;当前进程仍继续使用内存中的校准值。
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
function validEntry(entry) {
|
|
67
|
+
return !!entry
|
|
68
|
+
&& Number.isFinite(entry.correction)
|
|
69
|
+
&& entry.correction >= MIN_CORRECTION
|
|
70
|
+
&& entry.correction <= MAX_CORRECTION
|
|
71
|
+
&& Number.isInteger(entry.samples)
|
|
72
|
+
&& entry.samples > 0;
|
|
73
|
+
}
|
|
74
|
+
/** 读取指定 provider/model/工具集合的历史校准;未命中时退回 1。 */
|
|
75
|
+
export function getTokenCalibration(baseURL, model, tools) {
|
|
76
|
+
const entry = readCache().entries[calibrationKey(baseURL, model, tools)];
|
|
77
|
+
return validEntry(entry)
|
|
78
|
+
? { correction: entry.correction, samples: entry.samples }
|
|
79
|
+
: { correction: 1, samples: 0 };
|
|
80
|
+
}
|
|
81
|
+
/** 用一次真实 prompt usage 更新 EWMA;只落比例和样本数,不保存任何消息内容。 */
|
|
82
|
+
export function updateTokenCalibration(baseURL, model, tools, estimatedTokens, actualTokens) {
|
|
83
|
+
if (estimatedTokens <= 100
|
|
84
|
+
|| actualTokens <= 100
|
|
85
|
+
|| !Number.isFinite(estimatedTokens)
|
|
86
|
+
|| !Number.isFinite(actualTokens)) {
|
|
87
|
+
return getTokenCalibration(baseURL, model, tools);
|
|
88
|
+
}
|
|
89
|
+
const key = calibrationKey(baseURL, model, tools);
|
|
90
|
+
const store = readCache();
|
|
91
|
+
const previous = store.entries[key];
|
|
92
|
+
const raw = Math.max(MIN_CORRECTION, Math.min(MAX_CORRECTION, actualTokens / estimatedTokens));
|
|
93
|
+
const correction = validEntry(previous)
|
|
94
|
+
? previous.correction * (1 - EWMA_ALPHA) + raw * EWMA_ALPHA
|
|
95
|
+
: raw;
|
|
96
|
+
const next = {
|
|
97
|
+
correction,
|
|
98
|
+
samples: validEntry(previous) ? previous.samples + 1 : 1,
|
|
99
|
+
updatedAt: Date.now(),
|
|
100
|
+
};
|
|
101
|
+
store.entries[key] = next;
|
|
102
|
+
writeCache();
|
|
103
|
+
return { correction: next.correction, samples: next.samples };
|
|
104
|
+
}
|
package/dist/llm/index.js
CHANGED
|
@@ -2,6 +2,7 @@ import OpenAI from 'openai';
|
|
|
2
2
|
import { config } from '../config/index.js';
|
|
3
3
|
import { tools } from '../tools/registry.js';
|
|
4
4
|
import { getPlanDisabledTools } from '../tools/constants.js';
|
|
5
|
+
import { ThinkTagFilter } from './think-filter.js';
|
|
5
6
|
/**
|
|
6
7
|
* LLM 调用重试策略:
|
|
7
8
|
* 可重试 → 429 (rate limit) / 5xx (server) / APIConnectionError / Node 网络错 (ETIMEDOUT 等)
|
|
@@ -22,9 +23,6 @@ const RETRY_JITTER = 0.2;
|
|
|
22
23
|
* 与 OpenAI 兼容协议的独立 `reasoning_content` 字段不同,这些模型把 thinking 直接嵌进 content
|
|
23
24
|
* 字符串,期间不调 onText(spinner 持续转 ⠹ 思考中…),也不写入可见 content(history 不被思考段污染)。
|
|
24
25
|
*/
|
|
25
|
-
// 用 \u003c 表示 <,绕开本工具对 < 的处理(直接写 '<\u003cthink\u003e' 里 < 会被吃掉)。
|
|
26
|
-
const THINK_OPEN = '<think>';
|
|
27
|
-
const THINK_CLOSE = '</think>';
|
|
28
26
|
let client = new OpenAI({
|
|
29
27
|
baseURL: config.baseURL,
|
|
30
28
|
apiKey: config.apiKey,
|
|
@@ -276,15 +274,20 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
|
|
|
276
274
|
...(config.maxTokens ? { max_tokens: config.maxTokens } : {}),
|
|
277
275
|
...(config.includeUsage ? { stream_options: { include_usage: true } } : {}),
|
|
278
276
|
}, signal ? { signal } : undefined);
|
|
279
|
-
//
|
|
280
|
-
//
|
|
281
|
-
// 普通字符,buf 末尾为当前态保留 (label.length - 1) 个字符给下一 chunk 看。
|
|
277
|
+
// content 内嵌 think 标签由独立增量状态机过滤。它只暂存“可能组成标签”的后缀,
|
|
278
|
+
// 因而既能覆盖标签任意位置跨 chunk,也不会让普通正文固定延迟数个字符。
|
|
282
279
|
let visibleContent = '';
|
|
283
280
|
let consumedAny = false;
|
|
284
|
-
|
|
285
|
-
let buf = '';
|
|
281
|
+
const thinkFilter = new ThinkTagFilter();
|
|
286
282
|
let usage;
|
|
287
283
|
const toolAcc = new Map();
|
|
284
|
+
const emitVisible = (text) => {
|
|
285
|
+
if (!text)
|
|
286
|
+
return;
|
|
287
|
+
visibleContent += text;
|
|
288
|
+
handlers.onText?.(text);
|
|
289
|
+
consumedAny = true;
|
|
290
|
+
};
|
|
288
291
|
for await (const chunk of stream) {
|
|
289
292
|
// usage:末尾 chunk(choices 可能为空)在 include_usage 时携带;先读再 continue。
|
|
290
293
|
if (chunk.usage) {
|
|
@@ -300,78 +303,12 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
|
|
|
300
303
|
const delta = chunk.choices?.[0]?.delta;
|
|
301
304
|
if (!delta)
|
|
302
305
|
continue; // 末尾 usage-only chunk 等无 delta
|
|
303
|
-
if (delta.content)
|
|
304
|
-
|
|
305
|
-
// inThink 内丢弃;标签起始/闭合用 indexOf 在 buf 里扫描。
|
|
306
|
-
// 末尾预留 (label.length - 1) 字符给下一 chunk 防切分误判。
|
|
307
|
-
buf += delta.content;
|
|
308
|
-
let i = 0;
|
|
309
|
-
while (true) {
|
|
310
|
-
if (inThink) {
|
|
311
|
-
// 思考段内,扫描 THINK_CLOSE;末尾预留 THINK_CLOSE.length - 1 防跨 chunk 切分
|
|
312
|
-
const endIdx = buf.indexOf(THINK_CLOSE, i);
|
|
313
|
-
if (endIdx === -1) {
|
|
314
|
-
// 思考段内未找到闭合;buf 短到不可能包含 </think> 时全丢(都是思考段内容),
|
|
315
|
-
// 否则留 (THINK_CLOSE.length - 1) 给下一 chunk 防跨边界切分。
|
|
316
|
-
const safeLen = buf.length >= THINK_CLOSE.length
|
|
317
|
-
? buf.length - (THINK_CLOSE.length - 1)
|
|
318
|
-
: buf.length;
|
|
319
|
-
i = safeLen;
|
|
320
|
-
break;
|
|
321
|
-
}
|
|
322
|
-
inThink = false;
|
|
323
|
-
i = endIdx + THINK_CLOSE.length;
|
|
324
|
-
}
|
|
325
|
-
else {
|
|
326
|
-
// 普通段,扫描 THINK_OPEN;末尾预留 THINK_OPEN.length - 1 防跨 chunk 切分
|
|
327
|
-
const startIdx = buf.indexOf(THINK_OPEN, i);
|
|
328
|
-
if (startIdx === -1) {
|
|
329
|
-
// buf 短到不可能包含 <think> 时全输出(无 think 标签的普通模型不受影响);
|
|
330
|
-
// 否则留 (THINK_OPEN.length - 1) 给下一 chunk 防跨边界切分误判。
|
|
331
|
-
const safeLen = buf.length >= THINK_OPEN.length
|
|
332
|
-
? buf.length - (THINK_OPEN.length - 1)
|
|
333
|
-
: buf.length;
|
|
334
|
-
const seg = buf.slice(i, safeLen);
|
|
335
|
-
if (seg) {
|
|
336
|
-
visibleContent += seg;
|
|
337
|
-
handlers.onText?.(seg);
|
|
338
|
-
consumedAny = true;
|
|
339
|
-
}
|
|
340
|
-
i = safeLen;
|
|
341
|
-
break;
|
|
342
|
-
}
|
|
343
|
-
// THINK_OPEN 之前的普通段:输出
|
|
344
|
-
if (startIdx > i) {
|
|
345
|
-
const seg = buf.slice(i, startIdx);
|
|
346
|
-
visibleContent += seg;
|
|
347
|
-
handlers.onText?.(seg);
|
|
348
|
-
consumedAny = true;
|
|
349
|
-
}
|
|
350
|
-
inThink = true;
|
|
351
|
-
i = startIdx + THINK_OPEN.length;
|
|
352
|
-
}
|
|
353
|
-
}
|
|
354
|
-
buf = buf.slice(i);
|
|
355
|
-
}
|
|
306
|
+
if (delta.content)
|
|
307
|
+
emitVisible(thinkFilter.push(delta.content));
|
|
356
308
|
if (delta.tool_calls) {
|
|
357
|
-
//
|
|
358
|
-
//
|
|
359
|
-
//
|
|
360
|
-
// spinner,用户看到「话没说完就去调工具」——history 完整但屏幕渲染顺序错位。
|
|
361
|
-
// inThink 段照旧丢弃(思考中模型不会同时吐 tool_call,理论上 buf 不会有思考段);
|
|
362
|
-
// 防御性保留 !inThink 判断。
|
|
363
|
-
if (buf && !inThink) {
|
|
364
|
-
visibleContent += buf;
|
|
365
|
-
// 给 onText 渲染时剥掉尾部 \n:md 渲染器(contentWriteMd)把尾部 \n 当段落分隔 → 产空行;
|
|
366
|
-
// 随后 onToolCall 检测到 lastChar !== '\n' 会经 contentWrite('\n') 补一个原始换行
|
|
367
|
-
// (不走 md,只是普通行分隔,无空行)—— 与改造前 onToolCall 补 \n 的行为一致。
|
|
368
|
-
// visibleContent 保留原 buf(含 \n),history 完整不受影响。
|
|
369
|
-
const tail = buf.replace(/\n+$/, '');
|
|
370
|
-
if (tail)
|
|
371
|
-
handlers.onText?.(tail);
|
|
372
|
-
consumedAny = true;
|
|
373
|
-
buf = '';
|
|
374
|
-
}
|
|
309
|
+
// 不在这里 flush thinkFilter:其内部若有残留,只可能是 `<th` / `</thi` 一类
|
|
310
|
+
// 潜在标签前缀。旧实现把这段在工具转折点强制送进 onText,正是 `k>` 等残片
|
|
311
|
+
// 偶发混到工具摘要附近的来源。普通文本不会被状态机滞留。
|
|
375
312
|
for (const tc of delta.tool_calls) {
|
|
376
313
|
const idx = tc.index ?? 0;
|
|
377
314
|
let entry = toolAcc.get(idx);
|
|
@@ -392,19 +329,8 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
|
|
|
392
329
|
}
|
|
393
330
|
}
|
|
394
331
|
}
|
|
395
|
-
//
|
|
396
|
-
|
|
397
|
-
// (之前注释说"不再调 onText"是 bug——安全尾里的真实文本会被屏幕吞掉,用户看到模型
|
|
398
|
-
// 话没说完就去调工具 / 直接结束;history 有但显示缺。现在补上 onText 让屏幕与 history 一致。)
|
|
399
|
-
// - 思考段未闭合:丢弃,防 thinking 文本泄漏到 history
|
|
400
|
-
if (buf) {
|
|
401
|
-
if (!inThink) {
|
|
402
|
-
visibleContent += buf;
|
|
403
|
-
handlers.onText?.(buf);
|
|
404
|
-
consumedAny = true;
|
|
405
|
-
}
|
|
406
|
-
buf = '';
|
|
407
|
-
}
|
|
332
|
+
// 流结束后只释放普通态下真实的文本尾;未闭合思考段继续丢弃。
|
|
333
|
+
emitVisible(thinkFilter.finish());
|
|
408
334
|
const toolCalls = [...toolAcc.entries()]
|
|
409
335
|
.sort((a, b) => a[0] - b[0])
|
|
410
336
|
.map(([, e]) => ({
|
|
@@ -510,11 +436,28 @@ export function estimateMessagesTokens(messages) {
|
|
|
510
436
|
sum += messageTokens(m);
|
|
511
437
|
return sum;
|
|
512
438
|
}
|
|
513
|
-
|
|
514
|
-
/**
|
|
515
|
-
export function estimateToolSchemaTokens() {
|
|
516
|
-
if (
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
439
|
+
const schemaTokensCache = new WeakMap();
|
|
440
|
+
/** 估算本次实际发送的工具 schema token;按工具数组实例缓存。 */
|
|
441
|
+
export function estimateToolSchemaTokens(activeTools = chatTools) {
|
|
442
|
+
if (activeTools.length === 0)
|
|
443
|
+
return 0;
|
|
444
|
+
const key = activeTools;
|
|
445
|
+
const cached = schemaTokensCache.get(key);
|
|
446
|
+
if (cached !== undefined)
|
|
447
|
+
return cached;
|
|
448
|
+
const estimated = estimateTokens(JSON.stringify(activeTools)) + 16;
|
|
449
|
+
schemaTokensCache.set(key, estimated);
|
|
450
|
+
return estimated;
|
|
451
|
+
}
|
|
452
|
+
/** 把模型级校正系数统一应用到原始估算。 */
|
|
453
|
+
export function correctTokenEstimate(estimate, correction = 1) {
|
|
454
|
+
const safeCorrection = Number.isFinite(correction)
|
|
455
|
+
? Math.max(0.5, Math.min(2, correction))
|
|
456
|
+
: 1;
|
|
457
|
+
return estimate > 0 ? Math.max(1, Math.ceil(estimate * safeCorrection)) : 0;
|
|
458
|
+
}
|
|
459
|
+
/** 估算一次完整请求的 prompt token(messages + 本次实际工具 schema)。 */
|
|
460
|
+
export function estimatePromptTokens(messages, activeTools = chatTools, correction = 1) {
|
|
461
|
+
const raw = estimateMessagesTokens(messages) + estimateToolSchemaTokens(activeTools);
|
|
462
|
+
return correctTokenEstimate(raw, correction);
|
|
520
463
|
}
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 增量过滤部分 OpenAI 兼容后端直接混在 content 中的 <think>...</think>。
|
|
3
|
+
*
|
|
4
|
+
* 关键点:流式 chunk 可以在标签任意字符间切开,因此不能仅在当前 chunk 内查找,
|
|
5
|
+
* 也不能把“短于标签”的缓冲直接输出。普通态还要吞掉孤立 </think>:部分后端把
|
|
6
|
+
* reasoning 放在独立字段,却仍在 content 的开头附带闭标签。
|
|
7
|
+
*/
|
|
8
|
+
const THINK_OPEN = '<think>';
|
|
9
|
+
const THINK_CLOSE = '</think>';
|
|
10
|
+
/** 返回 text 末尾与任一 tag 前缀重合的最长长度。 */
|
|
11
|
+
function trailingTagPrefixLength(text, tags) {
|
|
12
|
+
const max = Math.min(text.length, Math.max(...tags.map((tag) => tag.length - 1)));
|
|
13
|
+
for (let length = max; length > 0; length--) {
|
|
14
|
+
const suffix = text.slice(-length);
|
|
15
|
+
if (tags.some((tag) => tag.startsWith(suffix)))
|
|
16
|
+
return length;
|
|
17
|
+
}
|
|
18
|
+
return 0;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* 每次 push 返回当前已经能够确认是正文的文本;标签和思考内容永不返回。
|
|
22
|
+
* finish 必须在流结束时调用,以释放普通正文末尾暂存的 `<` 等潜在标签前缀。
|
|
23
|
+
*/
|
|
24
|
+
export class ThinkTagFilter {
|
|
25
|
+
buffer = '';
|
|
26
|
+
inThink = false;
|
|
27
|
+
push(chunk) {
|
|
28
|
+
if (!chunk)
|
|
29
|
+
return '';
|
|
30
|
+
this.buffer += chunk;
|
|
31
|
+
return this.drain(false);
|
|
32
|
+
}
|
|
33
|
+
finish() {
|
|
34
|
+
return this.drain(true);
|
|
35
|
+
}
|
|
36
|
+
drain(final) {
|
|
37
|
+
let visible = '';
|
|
38
|
+
while (this.buffer) {
|
|
39
|
+
if (this.inThink) {
|
|
40
|
+
const closeIdx = this.buffer.indexOf(THINK_CLOSE);
|
|
41
|
+
if (closeIdx >= 0) {
|
|
42
|
+
this.buffer = this.buffer.slice(closeIdx + THINK_CLOSE.length);
|
|
43
|
+
this.inThink = false;
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
if (final) {
|
|
47
|
+
// 未闭合思考段一直丢弃,不能在流结束时误当正文释放。
|
|
48
|
+
this.buffer = '';
|
|
49
|
+
break;
|
|
50
|
+
}
|
|
51
|
+
// 思考正文可立即丢弃,只保留可能跨 chunk 组成 </think> 的后缀。
|
|
52
|
+
const keep = trailingTagPrefixLength(this.buffer, [THINK_CLOSE]);
|
|
53
|
+
this.buffer = keep > 0 ? this.buffer.slice(-keep) : '';
|
|
54
|
+
break;
|
|
55
|
+
}
|
|
56
|
+
const openIdx = this.buffer.indexOf(THINK_OPEN);
|
|
57
|
+
const closeIdx = this.buffer.indexOf(THINK_CLOSE);
|
|
58
|
+
let tagIdx = -1;
|
|
59
|
+
let tag = '';
|
|
60
|
+
if (openIdx >= 0 && (closeIdx < 0 || openIdx < closeIdx)) {
|
|
61
|
+
tagIdx = openIdx;
|
|
62
|
+
tag = THINK_OPEN;
|
|
63
|
+
}
|
|
64
|
+
else if (closeIdx >= 0) {
|
|
65
|
+
// 独立 reasoning_content 后偶发残留的孤立闭标签也属于协议噪声。
|
|
66
|
+
tagIdx = closeIdx;
|
|
67
|
+
tag = THINK_CLOSE;
|
|
68
|
+
}
|
|
69
|
+
if (tagIdx >= 0) {
|
|
70
|
+
visible += this.buffer.slice(0, tagIdx);
|
|
71
|
+
this.buffer = this.buffer.slice(tagIdx + tag.length);
|
|
72
|
+
if (tag === THINK_OPEN)
|
|
73
|
+
this.inThink = true;
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
if (final) {
|
|
77
|
+
visible += this.buffer;
|
|
78
|
+
this.buffer = '';
|
|
79
|
+
break;
|
|
80
|
+
}
|
|
81
|
+
// 只暂存“确实可能成为标签”的后缀;普通文本立即输出,不引入固定 6/7 字符延迟。
|
|
82
|
+
const keep = trailingTagPrefixLength(this.buffer, [THINK_OPEN, THINK_CLOSE]);
|
|
83
|
+
const emitLength = this.buffer.length - keep;
|
|
84
|
+
visible += this.buffer.slice(0, emitLength);
|
|
85
|
+
this.buffer = this.buffer.slice(emitLength);
|
|
86
|
+
break;
|
|
87
|
+
}
|
|
88
|
+
return visible;
|
|
89
|
+
}
|
|
90
|
+
}
|
package/dist/repl/index.js
CHANGED
|
@@ -888,7 +888,6 @@ export async function startRepl(initialHistory, sessionId, updateNotice = null,
|
|
|
888
888
|
if (!loadSnapshots(loaded.id))
|
|
889
889
|
rebuildFromHistory(history);
|
|
890
890
|
contextState.lastUsage = undefined;
|
|
891
|
-
contextState.correction = 1;
|
|
892
891
|
contextState.lifecycleStats = undefined;
|
|
893
892
|
lastTurnUsage = undefined; // 续接:旧会话的 token 累计已无意义,清空等下轮覆写
|
|
894
893
|
layout.clearContent();
|
|
@@ -1024,7 +1023,6 @@ export async function startRepl(initialHistory, sessionId, updateNotice = null,
|
|
|
1024
1023
|
setCurrentSessionId(undefined, process.cwd()); // 同步清空 session/state
|
|
1025
1024
|
turnCount = 0; // 反思 cadence 重新计数
|
|
1026
1025
|
contextState.lastUsage = undefined;
|
|
1027
|
-
contextState.correction = 1;
|
|
1028
1026
|
contextState.lifecycleStats = undefined;
|
|
1029
1027
|
lastTurnUsage = undefined; // 清空旧轮的 token 累计
|
|
1030
1028
|
pendingAttachments = []; // 一并清空待发图片
|
package/dist/session/compact.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { chat,
|
|
1
|
+
import { chat, chatTools, correctTokenEstimate, estimatePromptTokens, estimateTokens, } from '../llm/index.js';
|
|
2
2
|
import { config } from '../config/index.js';
|
|
3
3
|
import { MAX_HISTORY_RESULT, MAX_MEMORY_RESULT, MAX_OLD_TOOL_STUB, MAX_SKILL_RESULT } from '../tools/constants.js';
|
|
4
4
|
import { ui } from '../ui/theme.js';
|
|
@@ -6,13 +6,11 @@ import { Spinner } from '../ui/spinner.js';
|
|
|
6
6
|
import * as layout from '../ui/layout.js';
|
|
7
7
|
import { pruneAfterCompaction } from '../rollback/index.js';
|
|
8
8
|
import { toText } from '../context/utils.js';
|
|
9
|
+
import { DEFAULT_BUDGET_POLICY } from '../context/budget.js';
|
|
9
10
|
export function createContextState() {
|
|
10
|
-
return { lastEstimate: 0, correction: 1 };
|
|
11
|
+
return { lastEstimate: 0, correction: 1, calibrationSamples: 0 };
|
|
11
12
|
}
|
|
12
|
-
export const contextState =
|
|
13
|
-
lastEstimate: 0,
|
|
14
|
-
correction: 1,
|
|
15
|
-
};
|
|
13
|
+
export const contextState = createContextState();
|
|
16
14
|
/** 中截:text 太长时保 head + 标记 + tail,总长 ≤ max。 */
|
|
17
15
|
export function truncateMid(text, max) {
|
|
18
16
|
if (text.length <= max)
|
|
@@ -240,20 +238,22 @@ async function defaultSummarize(older, focus) {
|
|
|
240
238
|
*/
|
|
241
239
|
export async function compactHistory(history, opts) {
|
|
242
240
|
const state = opts.contextState ?? contextState;
|
|
243
|
-
const
|
|
244
|
-
const estimateBefore =
|
|
241
|
+
const activeTools = opts.tools ?? chatTools;
|
|
242
|
+
const estimateBefore = estimatePromptTokens(history, activeTools, state.correction);
|
|
245
243
|
state.lastEstimate = estimateBefore;
|
|
246
244
|
const groups = groupFromEnd(history);
|
|
247
|
-
//
|
|
248
|
-
const keepBudget = Math.floor(opts.window *
|
|
245
|
+
// 保近期:按策略中的 token 比例累积(至少保 1 组),永不劈开 group。
|
|
246
|
+
const keepBudget = Math.floor(opts.window * DEFAULT_BUDGET_POLICY.compactKeepRatio);
|
|
249
247
|
const kept = [];
|
|
250
248
|
let keptTokens = 0;
|
|
251
249
|
for (let k = groups.length - 1; k >= 0; k--) {
|
|
252
250
|
const g = groups[k];
|
|
253
|
-
|
|
251
|
+
const nextTokens = keptTokens + groupTokens(g);
|
|
252
|
+
if (kept.length >= 1
|
|
253
|
+
&& correctTokenEstimate(nextTokens, state.correction) > keepBudget)
|
|
254
254
|
break;
|
|
255
255
|
kept.unshift(g);
|
|
256
|
-
keptTokens
|
|
256
|
+
keptTokens = nextTokens;
|
|
257
257
|
}
|
|
258
258
|
const oldGroups = groups.slice(0, groups.length - kept.length);
|
|
259
259
|
const noop = {
|
|
@@ -312,10 +312,9 @@ export async function compactHistory(history, opts) {
|
|
|
312
312
|
history.length = 0;
|
|
313
313
|
history.push(...rebuilt);
|
|
314
314
|
pruneAfterCompaction(history);
|
|
315
|
-
const estimateAfter =
|
|
315
|
+
const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
|
|
316
316
|
state.lastEstimate = estimateAfter;
|
|
317
317
|
state.lastUsage = undefined;
|
|
318
|
-
state.correction = 1;
|
|
319
318
|
layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}强制压缩(focus on early history)${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
|
|
320
319
|
return {
|
|
321
320
|
compacted: true,
|
|
@@ -400,10 +399,9 @@ export async function compactHistory(history, opts) {
|
|
|
400
399
|
history.length = 0;
|
|
401
400
|
history.push(...rebuilt);
|
|
402
401
|
pruneAfterCompaction(history); // 摘要删了旧轮次 → 按存活轮次裁剪回滚日志
|
|
403
|
-
const estimateAfter =
|
|
402
|
+
const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
|
|
404
403
|
state.lastEstimate = estimateAfter;
|
|
405
|
-
state.lastUsage = undefined; // 压缩后旧 usage 失效,/context
|
|
406
|
-
state.correction = 1;
|
|
404
|
+
state.lastUsage = undefined; // 压缩后旧 usage 失效,/context 改用校正估算
|
|
407
405
|
layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}压缩上下文${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
|
|
408
406
|
// 抖动保护:压缩后仍超阈 → 提示 /clear,不死循环
|
|
409
407
|
if (estimateAfter >= opts.threshold * opts.window) {
|
|
@@ -419,10 +417,9 @@ export async function compactHistory(history, opts) {
|
|
|
419
417
|
};
|
|
420
418
|
}
|
|
421
419
|
// 摘要失败:回退仅微压缩(tool content 已原地改),结构不动
|
|
422
|
-
const estimateAfter =
|
|
420
|
+
const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
|
|
423
421
|
state.lastEstimate = estimateAfter;
|
|
424
|
-
state.lastUsage = undefined; //
|
|
425
|
-
state.correction = 1;
|
|
422
|
+
state.lastUsage = undefined; // token 数已变,旧 usage 失效
|
|
426
423
|
if (microcompactDone) {
|
|
427
424
|
layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}微压缩旧工具结果${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
|
|
428
425
|
return {
|
|
@@ -460,9 +457,8 @@ export async function compactHistory(history, opts) {
|
|
|
460
457
|
* 强制走 compactHistory(manual/force 参数透传)。返 CompactResult 给 caller 文案展示。
|
|
461
458
|
* 默认 manual=false 自动路径完全不变。
|
|
462
459
|
*/
|
|
463
|
-
export async function maybeCompact(history, report, manualOpts, state = contextState) {
|
|
464
|
-
const
|
|
465
|
-
const est = estimateMessagesTokens(history) + schemaTokens;
|
|
460
|
+
export async function maybeCompact(history, report, manualOpts, state = contextState, activeTools = chatTools) {
|
|
461
|
+
const est = estimatePromptTokens(history, activeTools, state.correction);
|
|
466
462
|
state.lastEstimate = est;
|
|
467
463
|
const isManual = manualOpts?.manual === true;
|
|
468
464
|
// 手动路径:旁路 autoCompact / report / 总阈三重门
|
|
@@ -487,6 +483,7 @@ export async function maybeCompact(history, report, manualOpts, state = contextS
|
|
|
487
483
|
focus: manualOpts?.focus,
|
|
488
484
|
manual: isManual,
|
|
489
485
|
force: manualOpts?.force,
|
|
486
|
+
tools: activeTools,
|
|
490
487
|
contextState: state,
|
|
491
488
|
});
|
|
492
489
|
return r;
|
package/dist/session/index.js
CHANGED
|
@@ -6,9 +6,9 @@
|
|
|
6
6
|
*/
|
|
7
7
|
export { compactHistory, maybeCompact, capToolResultForHistory, truncateMid, contextState, createContextState, } from './compact.js';
|
|
8
8
|
// ── Context Budget Scheduler 接缝 ────────────────────────────────────────
|
|
9
|
-
// agent/core.ts
|
|
10
|
-
//
|
|
11
|
-
// repl /compact 命令调 manualCompact(history, focus?)
|
|
9
|
+
// agent/core.ts 在 age-aware sweep 后调用 runScheduler(history, step):评估五区预算,
|
|
10
|
+
// 只执行可落地的 warn / compact_history;开关关闭时退化为 maybeCompact。
|
|
11
|
+
// repl /compact 命令调 manualCompact(history, focus?):与自动路径共享决策,focus 透传摘要 prompt。
|
|
12
12
|
export { runScheduler, manualCompact, createBudgetScheduler, } from './scheduler.js';
|
|
13
13
|
export { dropContextFromHistory, formatDropResult, } from './drop.js';
|
|
14
14
|
export { newSessionId, saveSession, loadSession, listSessions, sessionDir, } from './persist.js';
|
|
@@ -1,68 +1,48 @@
|
|
|
1
|
-
// Context Budget Scheduler(执行层)
|
|
2
|
-
//
|
|
3
|
-
// 关系图:
|
|
1
|
+
// Context Budget Scheduler(执行层):评估五区预算并执行可落地的动作。
|
|
4
2
|
//
|
|
5
3
|
// agent/core.ts (步前) repl /compact 命令(手动)
|
|
6
4
|
// ↓ runScheduler(history, step) ↓ manualCompact(history, focus?)
|
|
7
5
|
// session/scheduler.ts(本文件)
|
|
8
6
|
// ├─ evaluateBudget(history, window, step) → BudgetReport
|
|
9
|
-
//
|
|
10
|
-
// ├─ scheduleActions(report) → ScheduleAction[]
|
|
11
|
-
// │ ↓
|
|
7
|
+
// ├─ scheduleActions(report) → warn | compact_history
|
|
12
8
|
// └─ 执行 actions:
|
|
13
|
-
// - warn:
|
|
14
|
-
// -
|
|
15
|
-
// - shrink_cold_tools L2: 由 pruner.observePush(relevance.ts)自动处理;此处 no-op
|
|
16
|
-
// - shrink_cold_tools L3: 由 lifecycle.pushTool(lifecycle.ts)自动处理;此处 no-op
|
|
17
|
-
// - cap_hot_tools: Hot 区只 cap,实际仍由 push-time cap 走;此处 no-op + 记日志
|
|
18
|
-
// - compact_history: 调 maybeCompact(history, report)── report 路由到 ROI 调度
|
|
9
|
+
// - warn: 仅写日志
|
|
10
|
+
// - compact_history: 调 maybeCompact(history, report)
|
|
19
11
|
//
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
//
|
|
23
|
-
// - **hotBoundary** 仍由调度器算出来供 lifecycle 内部用(将来可演进成「仅 Cold 区跑 age stub」)——
|
|
24
|
-
// 现版本先全面暴露给 report,暂不传参给 lifecycle。
|
|
25
|
-
// - **actionLog**:每次执行的决策落进 contextState.schedulerLog,供 /context 命令与调试用。
|
|
26
|
-
// - **manualCompact**:手动入口(用户敲 /compact),与 runScheduler 共享决策路径;唯一差别是
|
|
27
|
-
// 即使 history 不超预算也强制产 compact_history(focus 透传)。对齐用户拍板的方案 A。
|
|
12
|
+
// push-time cap / relevance / lifecycle / age-aware sweep 在进入 scheduler 前独立完成,
|
|
13
|
+
// scheduler 不再生成无法执行的 Cold/Hot tool action。每次决策写入 actionLog,供 /context 调试。
|
|
14
|
+
// manualCompact 与自动路径共享 scheduleActions,但用户显式触发时强制追加 compact_history。
|
|
28
15
|
//
|
|
29
16
|
// 开关:
|
|
30
17
|
// - config.contextBudget !== false(默认 true):agent 调 runScheduler
|
|
31
18
|
// - 关时 agent 仍走原 maybeCompact(history)无 report 路径,零行为变化
|
|
32
19
|
// - 手动 /compact 走 manualCompact;关时退化直接调 compactHistory(history, { focus })
|
|
33
20
|
import { evaluateBudget, scheduleActions, formatReport, } from '../context/budget.js';
|
|
21
|
+
import { chatTools } from '../llm/index.js';
|
|
34
22
|
import { config } from '../config/index.js';
|
|
35
23
|
import { maybeCompact, contextState } from './compact.js';
|
|
36
24
|
import * as layout from '../ui/layout.js';
|
|
37
25
|
import { ui } from '../ui/theme.js';
|
|
38
|
-
/** 创建 runAgentCore 闭包持有的 scheduler(每次 agent 启动一个新实例)。
|
|
39
|
-
* observePush 当前只是占位:真正 L1/L2/L3 已由 cap / pruner / lifecycle 在 push 时跑;
|
|
40
|
-
* 保留接口为后续「调度器注入 hotBoundary 给 lifecycle」演进留接缝。 */
|
|
26
|
+
/** 创建 runAgentCore 闭包持有的 scheduler(每次 agent 启动一个新实例)。 */
|
|
41
27
|
export function createBudgetScheduler(state = contextState) {
|
|
42
28
|
const obs = {
|
|
43
29
|
lastRunLog: null,
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
},
|
|
47
|
-
async runStep(history, step) {
|
|
48
|
-
const report = evaluateBudget(history, config.contextWindowTokens, step, state.correction);
|
|
30
|
+
async runStep(history, step, activeTools = chatTools) {
|
|
31
|
+
const report = evaluateBudget(history, config.contextWindowTokens, step, state.correction, activeTools);
|
|
49
32
|
const actions = scheduleActions(report);
|
|
50
33
|
let compactHistoryCalled = false;
|
|
51
34
|
let historyRebuilt = false;
|
|
52
35
|
for (const a of actions) {
|
|
53
36
|
if (a.kind === 'warn') {
|
|
54
37
|
// system 超:写一行提示(配置漂移应由用户处理,不是调度器压)
|
|
55
|
-
layout.contentWrite(` ${ui.yellow}●${ui.reset} ${ui.yellow}
|
|
38
|
+
layout.contentWrite(` ${ui.yellow}●${ui.reset} ${ui.yellow}调度器警告 [${a.layer}] ${a.reason}${ui.reset}\n`);
|
|
56
39
|
}
|
|
57
40
|
else if (a.kind === 'compact_history') {
|
|
58
41
|
// 路由到 maybeCompact;把结构重建信号传回 core,使 lifecycle 按新 index 恢复。
|
|
59
|
-
const result = await maybeCompact(history, report, undefined, state);
|
|
42
|
+
const result = await maybeCompact(history, report, undefined, state, activeTools);
|
|
60
43
|
compactHistoryCalled = true;
|
|
61
44
|
historyRebuilt ||= result?.historyRebuilt === true;
|
|
62
45
|
}
|
|
63
|
-
// shrink_cold_tools L1/L2/L3 与 cap_hot_tools:已由 push-time 闸在每次 push 自动跑
|
|
64
|
-
// (cap = MAX_HISTORY_RESULT;pruner = same-path 新旧替换;lifecycle = age stub)。
|
|
65
|
-
// 调度器不重复,只把决策记录下来供调试。
|
|
66
46
|
}
|
|
67
47
|
const log = {
|
|
68
48
|
step,
|
|
@@ -80,13 +60,13 @@ export function createBudgetScheduler(state = contextState) {
|
|
|
80
60
|
return obs;
|
|
81
61
|
}
|
|
82
62
|
/** 便捷:agent/core.ts 不需要每次 createBudgetScheduler,直接 runScheduler(history, step)。 */
|
|
83
|
-
export async function runScheduler(history, step, state = contextState) {
|
|
63
|
+
export async function runScheduler(history, step, state = contextState, activeTools = chatTools) {
|
|
84
64
|
const s = createBudgetScheduler(state);
|
|
85
|
-
return s.runStep(history, step);
|
|
65
|
+
return s.runStep(history, step, activeTools);
|
|
86
66
|
}
|
|
87
|
-
/** 手动 /compact 入口(repl)
|
|
88
|
-
* 即便 layers.history.overBudget=false 或 totalOver=false
|
|
89
|
-
*
|
|
67
|
+
/** 手动 /compact 入口(repl):与自动路径共享预算评估和可执行 action,但强制执行 history 摘要。
|
|
68
|
+
* 即便 layers.history.overBudget=false 或 totalOver=false,manual 仍追加 compact_history,
|
|
69
|
+
* 并把 focus 透传给 LLM 摘要 prompt。
|
|
90
70
|
*
|
|
91
71
|
* 关系:runScheduler 是「自动触发」,manualCompact 是「用户显式触发」,二者共享 scheduleActions。
|
|
92
72
|
*
|