mocode-ai 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,104 @@
1
+ import fs from 'node:fs';
2
+ import os from 'node:os';
3
+ import path from 'node:path';
4
+ import { createHash } from 'node:crypto';
5
+ const CACHE_VERSION = 1;
6
+ const EWMA_ALPHA = 0.2;
7
+ const MIN_CORRECTION = 0.5;
8
+ const MAX_CORRECTION = 2;
9
+ const MAX_ENTRIES = 64;
10
+ let cache;
11
+ const toolFingerprints = new WeakMap();
12
+ function cachePath() {
13
+ return process.env.MOCODE_TOKEN_CALIBRATION_CACHE
14
+ || path.join(os.homedir(), '.mocode', 'token-calibration.json');
15
+ }
16
+ function hash(value) {
17
+ return createHash('sha256').update(value).digest('hex');
18
+ }
19
+ function toolFingerprint(tools) {
20
+ const objectKey = tools;
21
+ const hit = toolFingerprints.get(objectKey);
22
+ if (hit)
23
+ return hit;
24
+ const fingerprint = hash(JSON.stringify(tools));
25
+ toolFingerprints.set(objectKey, fingerprint);
26
+ return fingerprint;
27
+ }
28
+ function calibrationKey(baseURL, model, tools) {
29
+ // 只把摘要写盘,避免 URL 中偶然携带的凭据出现在缓存文件。
30
+ return hash(`${baseURL}\0${model}\0${toolFingerprint(tools)}`);
31
+ }
32
+ function readCache() {
33
+ if (cache)
34
+ return cache;
35
+ try {
36
+ const parsed = JSON.parse(fs.readFileSync(cachePath(), 'utf8'));
37
+ if (parsed.version === CACHE_VERSION && parsed.entries && typeof parsed.entries === 'object') {
38
+ cache = parsed;
39
+ return cache;
40
+ }
41
+ }
42
+ catch {
43
+ // 不存在或损坏都从空缓存开始;校准是增强能力,不能阻断请求。
44
+ }
45
+ cache = { version: CACHE_VERSION, entries: {} };
46
+ return cache;
47
+ }
48
+ function writeCache() {
49
+ if (!cache)
50
+ return;
51
+ try {
52
+ const entries = Object.entries(cache.entries)
53
+ .sort(([, a], [, b]) => b.updatedAt - a.updatedAt)
54
+ .slice(0, MAX_ENTRIES);
55
+ cache.entries = Object.fromEntries(entries);
56
+ const target = cachePath();
57
+ fs.mkdirSync(path.dirname(target), { recursive: true });
58
+ const tmp = `${target}.tmp-${process.pid}`;
59
+ fs.writeFileSync(tmp, JSON.stringify(cache), 'utf8');
60
+ fs.renameSync(tmp, target);
61
+ }
62
+ catch {
63
+ // 只影响跨进程复用;当前进程仍继续使用内存中的校准值。
64
+ }
65
+ }
66
+ function validEntry(entry) {
67
+ return !!entry
68
+ && Number.isFinite(entry.correction)
69
+ && entry.correction >= MIN_CORRECTION
70
+ && entry.correction <= MAX_CORRECTION
71
+ && Number.isInteger(entry.samples)
72
+ && entry.samples > 0;
73
+ }
74
+ /** 读取指定 provider/model/工具集合的历史校准;未命中时退回 1。 */
75
+ export function getTokenCalibration(baseURL, model, tools) {
76
+ const entry = readCache().entries[calibrationKey(baseURL, model, tools)];
77
+ return validEntry(entry)
78
+ ? { correction: entry.correction, samples: entry.samples }
79
+ : { correction: 1, samples: 0 };
80
+ }
81
+ /** 用一次真实 prompt usage 更新 EWMA;只落比例和样本数,不保存任何消息内容。 */
82
+ export function updateTokenCalibration(baseURL, model, tools, estimatedTokens, actualTokens) {
83
+ if (estimatedTokens <= 100
84
+ || actualTokens <= 100
85
+ || !Number.isFinite(estimatedTokens)
86
+ || !Number.isFinite(actualTokens)) {
87
+ return getTokenCalibration(baseURL, model, tools);
88
+ }
89
+ const key = calibrationKey(baseURL, model, tools);
90
+ const store = readCache();
91
+ const previous = store.entries[key];
92
+ const raw = Math.max(MIN_CORRECTION, Math.min(MAX_CORRECTION, actualTokens / estimatedTokens));
93
+ const correction = validEntry(previous)
94
+ ? previous.correction * (1 - EWMA_ALPHA) + raw * EWMA_ALPHA
95
+ : raw;
96
+ const next = {
97
+ correction,
98
+ samples: validEntry(previous) ? previous.samples + 1 : 1,
99
+ updatedAt: Date.now(),
100
+ };
101
+ store.entries[key] = next;
102
+ writeCache();
103
+ return { correction: next.correction, samples: next.samples };
104
+ }
package/dist/llm/index.js CHANGED
@@ -2,6 +2,7 @@ import OpenAI from 'openai';
2
2
  import { config } from '../config/index.js';
3
3
  import { tools } from '../tools/registry.js';
4
4
  import { getPlanDisabledTools } from '../tools/constants.js';
5
+ import { ThinkTagFilter } from './think-filter.js';
5
6
  /**
6
7
  * LLM 调用重试策略:
7
8
  * 可重试 → 429 (rate limit) / 5xx (server) / APIConnectionError / Node 网络错 (ETIMEDOUT 等)
@@ -22,9 +23,6 @@ const RETRY_JITTER = 0.2;
22
23
  * 与 OpenAI 兼容协议的独立 `reasoning_content` 字段不同,这些模型把 thinking 直接嵌进 content
23
24
  * 字符串,期间不调 onText(spinner 持续转 ⠹ 思考中…),也不写入可见 content(history 不被思考段污染)。
24
25
  */
25
- // 用 \u003c 表示 <,绕开本工具对 < 的处理(直接写 '<\u003cthink\u003e' 里 < 会被吃掉)。
26
- const THINK_OPEN = '<think>';
27
- const THINK_CLOSE = '</think>';
28
26
  let client = new OpenAI({
29
27
  baseURL: config.baseURL,
30
28
  apiKey: config.apiKey,
@@ -276,15 +274,20 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
276
274
  ...(config.maxTokens ? { max_tokens: config.maxTokens } : {}),
277
275
  ...(config.includeUsage ? { stream_options: { include_usage: true } } : {}),
278
276
  }, signal ? { signal } : undefined);
279
- // 流式 start end 标签过滤(见模块顶部 THINK_OPEN/CLOSE)。
280
- // 跨 chunk 切分防御:buf 累积跨 chunk 边界,indexOf 扫描;为避免把跨 chunk 标签误判为
281
- // 普通字符,buf 末尾为当前态保留 (label.length - 1) 个字符给下一 chunk 看。
277
+ // content 内嵌 think 标签由独立增量状态机过滤。它只暂存“可能组成标签”的后缀,
278
+ // 因而既能覆盖标签任意位置跨 chunk,也不会让普通正文固定延迟数个字符。
282
279
  let visibleContent = '';
283
280
  let consumedAny = false;
284
- let inThink = false;
285
- let buf = '';
281
+ const thinkFilter = new ThinkTagFilter();
286
282
  let usage;
287
283
  const toolAcc = new Map();
284
+ const emitVisible = (text) => {
285
+ if (!text)
286
+ return;
287
+ visibleContent += text;
288
+ handlers.onText?.(text);
289
+ consumedAny = true;
290
+ };
288
291
  for await (const chunk of stream) {
289
292
  // usage:末尾 chunk(choices 可能为空)在 include_usage 时携带;先读再 continue。
290
293
  if (chunk.usage) {
@@ -300,78 +303,12 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
300
303
  const delta = chunk.choices?.[0]?.delta;
301
304
  if (!delta)
302
305
  continue; // 末尾 usage-only chunk 等无 delta
303
- if (delta.content) {
304
- // 状态机切分 delta.content:inThink 外输出到 visibleContent + onText;
305
- // inThink 内丢弃;标签起始/闭合用 indexOf 在 buf 里扫描。
306
- // 末尾预留 (label.length - 1) 字符给下一 chunk 防切分误判。
307
- buf += delta.content;
308
- let i = 0;
309
- while (true) {
310
- if (inThink) {
311
- // 思考段内,扫描 THINK_CLOSE;末尾预留 THINK_CLOSE.length - 1 防跨 chunk 切分
312
- const endIdx = buf.indexOf(THINK_CLOSE, i);
313
- if (endIdx === -1) {
314
- // 思考段内未找到闭合;buf 短到不可能包含 </think> 时全丢(都是思考段内容),
315
- // 否则留 (THINK_CLOSE.length - 1) 给下一 chunk 防跨边界切分。
316
- const safeLen = buf.length >= THINK_CLOSE.length
317
- ? buf.length - (THINK_CLOSE.length - 1)
318
- : buf.length;
319
- i = safeLen;
320
- break;
321
- }
322
- inThink = false;
323
- i = endIdx + THINK_CLOSE.length;
324
- }
325
- else {
326
- // 普通段,扫描 THINK_OPEN;末尾预留 THINK_OPEN.length - 1 防跨 chunk 切分
327
- const startIdx = buf.indexOf(THINK_OPEN, i);
328
- if (startIdx === -1) {
329
- // buf 短到不可能包含 <think> 时全输出(无 think 标签的普通模型不受影响);
330
- // 否则留 (THINK_OPEN.length - 1) 给下一 chunk 防跨边界切分误判。
331
- const safeLen = buf.length >= THINK_OPEN.length
332
- ? buf.length - (THINK_OPEN.length - 1)
333
- : buf.length;
334
- const seg = buf.slice(i, safeLen);
335
- if (seg) {
336
- visibleContent += seg;
337
- handlers.onText?.(seg);
338
- consumedAny = true;
339
- }
340
- i = safeLen;
341
- break;
342
- }
343
- // THINK_OPEN 之前的普通段:输出
344
- if (startIdx > i) {
345
- const seg = buf.slice(i, startIdx);
346
- visibleContent += seg;
347
- handlers.onText?.(seg);
348
- consumedAny = true;
349
- }
350
- inThink = true;
351
- i = startIdx + THINK_OPEN.length;
352
- }
353
- }
354
- buf = buf.slice(i);
355
- }
306
+ if (delta.content)
307
+ emitVisible(thinkFilter.push(delta.content));
356
308
  if (delta.tool_calls) {
357
- // 文本→工具转折点:若 buf 还残留普通文本的安全尾(为防跨 chunk 切分留的
358
- // THINK_OPEN.length - 1 字符),立即 flush 到屏幕。否则这段尾巴会一直搁置到
359
- // 流结束才由尾部防御输出,而那时 onToolCall 早已触发、TUI 已补换行+生成中
360
- // spinner,用户看到「话没说完就去调工具」——history 完整但屏幕渲染顺序错位。
361
- // inThink 段照旧丢弃(思考中模型不会同时吐 tool_call,理论上 buf 不会有思考段);
362
- // 防御性保留 !inThink 判断。
363
- if (buf && !inThink) {
364
- visibleContent += buf;
365
- // 给 onText 渲染时剥掉尾部 \n:md 渲染器(contentWriteMd)把尾部 \n 当段落分隔 → 产空行;
366
- // 随后 onToolCall 检测到 lastChar !== '\n' 会经 contentWrite('\n') 补一个原始换行
367
- // (不走 md,只是普通行分隔,无空行)—— 与改造前 onToolCall 补 \n 的行为一致。
368
- // visibleContent 保留原 buf(含 \n),history 完整不受影响。
369
- const tail = buf.replace(/\n+$/, '');
370
- if (tail)
371
- handlers.onText?.(tail);
372
- consumedAny = true;
373
- buf = '';
374
- }
309
+ // 不在这里 flush thinkFilter:其内部若有残留,只可能是 `<th` / `</thi` 一类
310
+ // 潜在标签前缀。旧实现把这段在工具转折点强制送进 onText,正是 `k>` 等残片
311
+ // 偶发混到工具摘要附近的来源。普通文本不会被状态机滞留。
375
312
  for (const tc of delta.tool_calls) {
376
313
  const idx = tc.index ?? 0;
377
314
  let entry = toolAcc.get(idx);
@@ -392,19 +329,8 @@ async function chatOnce(messages, handlers, signal, toolsOverride) {
392
329
  }
393
330
  }
394
331
  }
395
- // 防御:循环内 buf.slice 已把可确认部分消费;此处覆盖流末尾的"安全尾":
396
- // - 普通段(stream 已结束,标签不会再出现):作为可见内容追加到 visibleContent + 调 onText
397
- // (之前注释说"不再调 onText"是 bug——安全尾里的真实文本会被屏幕吞掉,用户看到模型
398
- // 话没说完就去调工具 / 直接结束;history 有但显示缺。现在补上 onText 让屏幕与 history 一致。)
399
- // - 思考段未闭合:丢弃,防 thinking 文本泄漏到 history
400
- if (buf) {
401
- if (!inThink) {
402
- visibleContent += buf;
403
- handlers.onText?.(buf);
404
- consumedAny = true;
405
- }
406
- buf = '';
407
- }
332
+ // 流结束后只释放普通态下真实的文本尾;未闭合思考段继续丢弃。
333
+ emitVisible(thinkFilter.finish());
408
334
  const toolCalls = [...toolAcc.entries()]
409
335
  .sort((a, b) => a[0] - b[0])
410
336
  .map(([, e]) => ({
@@ -510,11 +436,28 @@ export function estimateMessagesTokens(messages) {
510
436
  sum += messageTokens(m);
511
437
  return sum;
512
438
  }
513
- let schemaTokensCache;
514
- /** 估算 chatTools(工具 schema)占用的一次性 token,带缓存。 */
515
- export function estimateToolSchemaTokens() {
516
- if (schemaTokensCache === undefined) {
517
- schemaTokensCache = estimateTokens(JSON.stringify(chatTools)) + 16;
518
- }
519
- return schemaTokensCache;
439
+ const schemaTokensCache = new WeakMap();
440
+ /** 估算本次实际发送的工具 schema token;按工具数组实例缓存。 */
441
+ export function estimateToolSchemaTokens(activeTools = chatTools) {
442
+ if (activeTools.length === 0)
443
+ return 0;
444
+ const key = activeTools;
445
+ const cached = schemaTokensCache.get(key);
446
+ if (cached !== undefined)
447
+ return cached;
448
+ const estimated = estimateTokens(JSON.stringify(activeTools)) + 16;
449
+ schemaTokensCache.set(key, estimated);
450
+ return estimated;
451
+ }
452
+ /** 把模型级校正系数统一应用到原始估算。 */
453
+ export function correctTokenEstimate(estimate, correction = 1) {
454
+ const safeCorrection = Number.isFinite(correction)
455
+ ? Math.max(0.5, Math.min(2, correction))
456
+ : 1;
457
+ return estimate > 0 ? Math.max(1, Math.ceil(estimate * safeCorrection)) : 0;
458
+ }
459
+ /** 估算一次完整请求的 prompt token(messages + 本次实际工具 schema)。 */
460
+ export function estimatePromptTokens(messages, activeTools = chatTools, correction = 1) {
461
+ const raw = estimateMessagesTokens(messages) + estimateToolSchemaTokens(activeTools);
462
+ return correctTokenEstimate(raw, correction);
520
463
  }
@@ -0,0 +1,90 @@
1
+ /**
2
+ * 增量过滤部分 OpenAI 兼容后端直接混在 content 中的 <think>...</think>。
3
+ *
4
+ * 关键点:流式 chunk 可以在标签任意字符间切开,因此不能仅在当前 chunk 内查找,
5
+ * 也不能把“短于标签”的缓冲直接输出。普通态还要吞掉孤立 </think>:部分后端把
6
+ * reasoning 放在独立字段,却仍在 content 的开头附带闭标签。
7
+ */
8
+ const THINK_OPEN = '<think>';
9
+ const THINK_CLOSE = '</think>';
10
+ /** 返回 text 末尾与任一 tag 前缀重合的最长长度。 */
11
+ function trailingTagPrefixLength(text, tags) {
12
+ const max = Math.min(text.length, Math.max(...tags.map((tag) => tag.length - 1)));
13
+ for (let length = max; length > 0; length--) {
14
+ const suffix = text.slice(-length);
15
+ if (tags.some((tag) => tag.startsWith(suffix)))
16
+ return length;
17
+ }
18
+ return 0;
19
+ }
20
+ /**
21
+ * 每次 push 返回当前已经能够确认是正文的文本;标签和思考内容永不返回。
22
+ * finish 必须在流结束时调用,以释放普通正文末尾暂存的 `<` 等潜在标签前缀。
23
+ */
24
+ export class ThinkTagFilter {
25
+ buffer = '';
26
+ inThink = false;
27
+ push(chunk) {
28
+ if (!chunk)
29
+ return '';
30
+ this.buffer += chunk;
31
+ return this.drain(false);
32
+ }
33
+ finish() {
34
+ return this.drain(true);
35
+ }
36
+ drain(final) {
37
+ let visible = '';
38
+ while (this.buffer) {
39
+ if (this.inThink) {
40
+ const closeIdx = this.buffer.indexOf(THINK_CLOSE);
41
+ if (closeIdx >= 0) {
42
+ this.buffer = this.buffer.slice(closeIdx + THINK_CLOSE.length);
43
+ this.inThink = false;
44
+ continue;
45
+ }
46
+ if (final) {
47
+ // 未闭合思考段一直丢弃,不能在流结束时误当正文释放。
48
+ this.buffer = '';
49
+ break;
50
+ }
51
+ // 思考正文可立即丢弃,只保留可能跨 chunk 组成 </think> 的后缀。
52
+ const keep = trailingTagPrefixLength(this.buffer, [THINK_CLOSE]);
53
+ this.buffer = keep > 0 ? this.buffer.slice(-keep) : '';
54
+ break;
55
+ }
56
+ const openIdx = this.buffer.indexOf(THINK_OPEN);
57
+ const closeIdx = this.buffer.indexOf(THINK_CLOSE);
58
+ let tagIdx = -1;
59
+ let tag = '';
60
+ if (openIdx >= 0 && (closeIdx < 0 || openIdx < closeIdx)) {
61
+ tagIdx = openIdx;
62
+ tag = THINK_OPEN;
63
+ }
64
+ else if (closeIdx >= 0) {
65
+ // 独立 reasoning_content 后偶发残留的孤立闭标签也属于协议噪声。
66
+ tagIdx = closeIdx;
67
+ tag = THINK_CLOSE;
68
+ }
69
+ if (tagIdx >= 0) {
70
+ visible += this.buffer.slice(0, tagIdx);
71
+ this.buffer = this.buffer.slice(tagIdx + tag.length);
72
+ if (tag === THINK_OPEN)
73
+ this.inThink = true;
74
+ continue;
75
+ }
76
+ if (final) {
77
+ visible += this.buffer;
78
+ this.buffer = '';
79
+ break;
80
+ }
81
+ // 只暂存“确实可能成为标签”的后缀;普通文本立即输出,不引入固定 6/7 字符延迟。
82
+ const keep = trailingTagPrefixLength(this.buffer, [THINK_OPEN, THINK_CLOSE]);
83
+ const emitLength = this.buffer.length - keep;
84
+ visible += this.buffer.slice(0, emitLength);
85
+ this.buffer = this.buffer.slice(emitLength);
86
+ break;
87
+ }
88
+ return visible;
89
+ }
90
+ }
@@ -888,7 +888,6 @@ export async function startRepl(initialHistory, sessionId, updateNotice = null,
888
888
  if (!loadSnapshots(loaded.id))
889
889
  rebuildFromHistory(history);
890
890
  contextState.lastUsage = undefined;
891
- contextState.correction = 1;
892
891
  contextState.lifecycleStats = undefined;
893
892
  lastTurnUsage = undefined; // 续接:旧会话的 token 累计已无意义,清空等下轮覆写
894
893
  layout.clearContent();
@@ -1024,7 +1023,6 @@ export async function startRepl(initialHistory, sessionId, updateNotice = null,
1024
1023
  setCurrentSessionId(undefined, process.cwd()); // 同步清空 session/state
1025
1024
  turnCount = 0; // 反思 cadence 重新计数
1026
1025
  contextState.lastUsage = undefined;
1027
- contextState.correction = 1;
1028
1026
  contextState.lifecycleStats = undefined;
1029
1027
  lastTurnUsage = undefined; // 清空旧轮的 token 累计
1030
1028
  pendingAttachments = []; // 一并清空待发图片
@@ -1,4 +1,4 @@
1
- import { chat, estimateMessagesTokens, estimateToolSchemaTokens, estimateTokens, } from '../llm/index.js';
1
+ import { chat, chatTools, correctTokenEstimate, estimatePromptTokens, estimateTokens, } from '../llm/index.js';
2
2
  import { config } from '../config/index.js';
3
3
  import { MAX_HISTORY_RESULT, MAX_MEMORY_RESULT, MAX_OLD_TOOL_STUB, MAX_SKILL_RESULT } from '../tools/constants.js';
4
4
  import { ui } from '../ui/theme.js';
@@ -6,13 +6,11 @@ import { Spinner } from '../ui/spinner.js';
6
6
  import * as layout from '../ui/layout.js';
7
7
  import { pruneAfterCompaction } from '../rollback/index.js';
8
8
  import { toText } from '../context/utils.js';
9
+ import { DEFAULT_BUDGET_POLICY } from '../context/budget.js';
9
10
  export function createContextState() {
10
- return { lastEstimate: 0, correction: 1 };
11
+ return { lastEstimate: 0, correction: 1, calibrationSamples: 0 };
11
12
  }
12
- export const contextState = {
13
- lastEstimate: 0,
14
- correction: 1,
15
- };
13
+ export const contextState = createContextState();
16
14
  /** 中截:text 太长时保 head + 标记 + tail,总长 ≤ max。 */
17
15
  export function truncateMid(text, max) {
18
16
  if (text.length <= max)
@@ -240,20 +238,22 @@ async function defaultSummarize(older, focus) {
240
238
  */
241
239
  export async function compactHistory(history, opts) {
242
240
  const state = opts.contextState ?? contextState;
243
- const schemaTokens = estimateToolSchemaTokens();
244
- const estimateBefore = estimateMessagesTokens(history) + schemaTokens;
241
+ const activeTools = opts.tools ?? chatTools;
242
+ const estimateBefore = estimatePromptTokens(history, activeTools, state.correction);
245
243
  state.lastEstimate = estimateBefore;
246
244
  const groups = groupFromEnd(history);
247
- // 保近期:从尾向前累积直到预算花完(至少保 1 组),永不劈开 group。
248
- const keepBudget = Math.floor(opts.window * 0.4);
245
+ // 保近期:按策略中的 token 比例累积(至少保 1 组),永不劈开 group。
246
+ const keepBudget = Math.floor(opts.window * DEFAULT_BUDGET_POLICY.compactKeepRatio);
249
247
  const kept = [];
250
248
  let keptTokens = 0;
251
249
  for (let k = groups.length - 1; k >= 0; k--) {
252
250
  const g = groups[k];
253
- if (kept.length >= 1 && keptTokens + groupTokens(g) > keepBudget)
251
+ const nextTokens = keptTokens + groupTokens(g);
252
+ if (kept.length >= 1
253
+ && correctTokenEstimate(nextTokens, state.correction) > keepBudget)
254
254
  break;
255
255
  kept.unshift(g);
256
- keptTokens += groupTokens(g);
256
+ keptTokens = nextTokens;
257
257
  }
258
258
  const oldGroups = groups.slice(0, groups.length - kept.length);
259
259
  const noop = {
@@ -312,10 +312,9 @@ export async function compactHistory(history, opts) {
312
312
  history.length = 0;
313
313
  history.push(...rebuilt);
314
314
  pruneAfterCompaction(history);
315
- const estimateAfter = estimateMessagesTokens(history) + schemaTokens;
315
+ const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
316
316
  state.lastEstimate = estimateAfter;
317
317
  state.lastUsage = undefined;
318
- state.correction = 1;
319
318
  layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}强制压缩(focus on early history)${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
320
319
  return {
321
320
  compacted: true,
@@ -400,10 +399,9 @@ export async function compactHistory(history, opts) {
400
399
  history.length = 0;
401
400
  history.push(...rebuilt);
402
401
  pruneAfterCompaction(history); // 摘要删了旧轮次 → 按存活轮次裁剪回滚日志
403
- const estimateAfter = estimateMessagesTokens(history) + schemaTokens;
402
+ const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
404
403
  state.lastEstimate = estimateAfter;
405
- state.lastUsage = undefined; // 压缩后旧 usage 失效,/context 改用估算
406
- state.correction = 1;
404
+ state.lastUsage = undefined; // 压缩后旧 usage 失效,/context 改用校正估算
407
405
  layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}压缩上下文${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
408
406
  // 抖动保护:压缩后仍超阈 → 提示 /clear,不死循环
409
407
  if (estimateAfter >= opts.threshold * opts.window) {
@@ -419,10 +417,9 @@ export async function compactHistory(history, opts) {
419
417
  };
420
418
  }
421
419
  // 摘要失败:回退仅微压缩(tool content 已原地改),结构不动
422
- const estimateAfter = estimateMessagesTokens(history) + schemaTokens;
420
+ const estimateAfter = estimatePromptTokens(history, activeTools, state.correction);
423
421
  state.lastEstimate = estimateAfter;
424
- state.lastUsage = undefined; // 结构虽未变,但 token 数已变,旧 usage 失效
425
- state.correction = 1;
422
+ state.lastUsage = undefined; // token 数已变,旧 usage 失效
426
423
  if (microcompactDone) {
427
424
  layout.contentWrite(` ${ui.bold}${ui.accent}●${ui.reset} ${ui.accent}微压缩旧工具结果${ui.reset} ${ui.dim}${estimateBefore} → ${estimateAfter} tokens${ui.reset}\n`);
428
425
  return {
@@ -460,9 +457,8 @@ export async function compactHistory(history, opts) {
460
457
  * 强制走 compactHistory(manual/force 参数透传)。返 CompactResult 给 caller 文案展示。
461
458
  * 默认 manual=false 自动路径完全不变。
462
459
  */
463
- export async function maybeCompact(history, report, manualOpts, state = contextState) {
464
- const schemaTokens = estimateToolSchemaTokens();
465
- const est = estimateMessagesTokens(history) + schemaTokens;
460
+ export async function maybeCompact(history, report, manualOpts, state = contextState, activeTools = chatTools) {
461
+ const est = estimatePromptTokens(history, activeTools, state.correction);
466
462
  state.lastEstimate = est;
467
463
  const isManual = manualOpts?.manual === true;
468
464
  // 手动路径:旁路 autoCompact / report / 总阈三重门
@@ -487,6 +483,7 @@ export async function maybeCompact(history, report, manualOpts, state = contextS
487
483
  focus: manualOpts?.focus,
488
484
  manual: isManual,
489
485
  force: manualOpts?.force,
486
+ tools: activeTools,
490
487
  contextState: state,
491
488
  });
492
489
  return r;
@@ -6,9 +6,9 @@
6
6
  */
7
7
  export { compactHistory, maybeCompact, capToolResultForHistory, truncateMid, contextState, createContextState, } from './compact.js';
8
8
  // ── Context Budget Scheduler 接缝 ────────────────────────────────────────
9
- // agent/core.ts 步前调 runScheduler(history, step):评估五区预算 → 按 ROI 调度
10
- // shrink_cold_tools / cap_hot_tools / compact_history。开关关闭时退化为 maybeCompact。
11
- // repl /compact 命令调 manualCompact(history, focus?):与自动路径完全一致,focus 透传摘要 prompt。
9
+ // agent/core.ts 在 age-aware sweep 后调用 runScheduler(history, step):评估五区预算,
10
+ // 只执行可落地的 warn / compact_history;开关关闭时退化为 maybeCompact。
11
+ // repl /compact 命令调 manualCompact(history, focus?):与自动路径共享决策,focus 透传摘要 prompt。
12
12
  export { runScheduler, manualCompact, createBudgetScheduler, } from './scheduler.js';
13
13
  export { dropContextFromHistory, formatDropResult, } from './drop.js';
14
14
  export { newSessionId, saveSession, loadSession, listSessions, sessionDir, } from './persist.js';
@@ -1,68 +1,48 @@
1
- // Context Budget Scheduler(执行层):把 scheduleActions() 产生的动作落到既有闸上。
2
- //
3
- // 关系图:
1
+ // Context Budget Scheduler(执行层):评估五区预算并执行可落地的动作。
4
2
  //
5
3
  // agent/core.ts (步前) repl /compact 命令(手动)
6
4
  // ↓ runScheduler(history, step) ↓ manualCompact(history, focus?)
7
5
  // session/scheduler.ts(本文件)
8
6
  // ├─ evaluateBudget(history, window, step) → BudgetReport
9
- // │ ↓
10
- // ├─ scheduleActions(report) → ScheduleAction[]
11
- // │ ↓
7
+ // ├─ scheduleActions(report) → warn | compact_history
12
8
  // └─ 执行 actions:
13
- // - warn: 仅写日志
14
- // - shrink_cold_tools L1: 由 push-time cap.ts(MAX_HISTORY_RESULT)自动处理;此处 no-op
15
- // - shrink_cold_tools L2: 由 pruner.observePush(relevance.ts)自动处理;此处 no-op
16
- // - shrink_cold_tools L3: 由 lifecycle.pushTool(lifecycle.ts)自动处理;此处 no-op
17
- // - cap_hot_tools: Hot 区只 cap,实际仍由 push-time cap 走;此处 no-op + 记日志
18
- // - compact_history: 调 maybeCompact(history, report)── report 路由到 ROI 调度
9
+ // - warn: 仅写日志
10
+ // - compact_history: 调 maybeCompact(history, report)
19
11
  //
20
- // 设计意图:
21
- // - **L1/L2/L3 不重复实现**——push-time 已经自动跑过这三级。再在调度器做一遍是 Double-Action
22
- // 且破坏「调度器永不抛错 + 幂等」契约。调度器只负责"决策时点",push-time 闸负责"执行"。
23
- // - **hotBoundary** 仍由调度器算出来供 lifecycle 内部用(将来可演进成「仅 Cold 区跑 age stub」)——
24
- // 现版本先全面暴露给 report,暂不传参给 lifecycle。
25
- // - **actionLog**:每次执行的决策落进 contextState.schedulerLog,供 /context 命令与调试用。
26
- // - **manualCompact**:手动入口(用户敲 /compact),与 runScheduler 共享决策路径;唯一差别是
27
- // 即使 history 不超预算也强制产 compact_history(focus 透传)。对齐用户拍板的方案 A。
12
+ // push-time cap / relevance / lifecycle / age-aware sweep 在进入 scheduler 前独立完成,
13
+ // scheduler 不再生成无法执行的 Cold/Hot tool action。每次决策写入 actionLog,供 /context 调试。
14
+ // manualCompact 与自动路径共享 scheduleActions,但用户显式触发时强制追加 compact_history。
28
15
  //
29
16
  // 开关:
30
17
  // - config.contextBudget !== false(默认 true):agent 调 runScheduler
31
18
  // - 关时 agent 仍走原 maybeCompact(history)无 report 路径,零行为变化
32
19
  // - 手动 /compact 走 manualCompact;关时退化直接调 compactHistory(history, { focus })
33
20
  import { evaluateBudget, scheduleActions, formatReport, } from '../context/budget.js';
21
+ import { chatTools } from '../llm/index.js';
34
22
  import { config } from '../config/index.js';
35
23
  import { maybeCompact, contextState } from './compact.js';
36
24
  import * as layout from '../ui/layout.js';
37
25
  import { ui } from '../ui/theme.js';
38
- /** 创建 runAgentCore 闭包持有的 scheduler(每次 agent 启动一个新实例)。
39
- * observePush 当前只是占位:真正 L1/L2/L3 已由 cap / pruner / lifecycle 在 push 时跑;
40
- * 保留接口为后续「调度器注入 hotBoundary 给 lifecycle」演进留接缝。 */
26
+ /** 创建 runAgentCore 闭包持有的 scheduler(每次 agent 启动一个新实例)。 */
41
27
  export function createBudgetScheduler(state = contextState) {
42
28
  const obs = {
43
29
  lastRunLog: null,
44
- observePush(_history, _idx) {
45
- // 占位:push-time 三闸(cap / pruner / lifecycle)已自动跑;此接缝供将来演进。
46
- },
47
- async runStep(history, step) {
48
- const report = evaluateBudget(history, config.contextWindowTokens, step, state.correction);
30
+ async runStep(history, step, activeTools = chatTools) {
31
+ const report = evaluateBudget(history, config.contextWindowTokens, step, state.correction, activeTools);
49
32
  const actions = scheduleActions(report);
50
33
  let compactHistoryCalled = false;
51
34
  let historyRebuilt = false;
52
35
  for (const a of actions) {
53
36
  if (a.kind === 'warn') {
54
37
  // system 超:写一行提示(配置漂移应由用户处理,不是调度器压)
55
- layout.contentWrite(` ${ui.yellow}●${ui.reset} ${ui.yellow}调度器警告:${a.layer} ${a.reason}${ui.reset}\n`);
38
+ layout.contentWrite(` ${ui.yellow}●${ui.reset} ${ui.yellow}调度器警告 [${a.layer}] ${a.reason}${ui.reset}\n`);
56
39
  }
57
40
  else if (a.kind === 'compact_history') {
58
41
  // 路由到 maybeCompact;把结构重建信号传回 core,使 lifecycle 按新 index 恢复。
59
- const result = await maybeCompact(history, report, undefined, state);
42
+ const result = await maybeCompact(history, report, undefined, state, activeTools);
60
43
  compactHistoryCalled = true;
61
44
  historyRebuilt ||= result?.historyRebuilt === true;
62
45
  }
63
- // shrink_cold_tools L1/L2/L3 与 cap_hot_tools:已由 push-time 闸在每次 push 自动跑
64
- // (cap = MAX_HISTORY_RESULT;pruner = same-path 新旧替换;lifecycle = age stub)。
65
- // 调度器不重复,只把决策记录下来供调试。
66
46
  }
67
47
  const log = {
68
48
  step,
@@ -80,13 +60,13 @@ export function createBudgetScheduler(state = contextState) {
80
60
  return obs;
81
61
  }
82
62
  /** 便捷:agent/core.ts 不需要每次 createBudgetScheduler,直接 runScheduler(history, step)。 */
83
- export async function runScheduler(history, step, state = contextState) {
63
+ export async function runScheduler(history, step, state = contextState, activeTools = chatTools) {
84
64
  const s = createBudgetScheduler(state);
85
- return s.runStep(history, step);
65
+ return s.runStep(history, step, activeTools);
86
66
  }
87
- /** 手动 /compact 入口(repl):与自动路径完全一致——五区 ROI 调度,但 history 摘要强制执行。
88
- * 即便 layers.history.overBudget=false 或 totalOver=false,manual 仍产 compact_history action
89
- * 把 focus 透传给 LLM 摘要 prompt。其它 ROI 决策(cold tools / cap hot / warn)按 scheduleActions 走。
67
+ /** 手动 /compact 入口(repl):与自动路径共享预算评估和可执行 action,但强制执行 history 摘要。
68
+ * 即便 layers.history.overBudget=false 或 totalOver=false,manual 仍追加 compact_history,
69
+ * 并把 focus 透传给 LLM 摘要 prompt。
90
70
  *
91
71
  * 关系:runScheduler 是「自动触发」,manualCompact 是「用户显式触发」,二者共享 scheduleActions。
92
72
  *