mocode-ai 1.1.6 → 1.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +23 -37
  2. package/README.zh-CN.md +46 -38
  3. package/dist/agent/core.js +106 -395
  4. package/dist/agent/index.js +2 -21
  5. package/dist/agent/spawn.js +3 -5
  6. package/dist/agent/work-discipline.js +28 -82
  7. package/dist/config/index.js +7 -7
  8. package/dist/context/age-aware.js +18 -48
  9. package/dist/context/artifacts.js +19 -17
  10. package/dist/context/budget.js +27 -28
  11. package/dist/context/classifier.js +0 -1
  12. package/dist/context/encoders/index.js +4 -11
  13. package/dist/context/index.js +4 -7
  14. package/dist/context/lifecycle.js +115 -483
  15. package/dist/context/pipeline.js +8 -15
  16. package/dist/context/relevance.js +77 -55
  17. package/dist/host/stdio.js +0 -6
  18. package/dist/i18n/index.js +0 -6
  19. package/dist/index.js +11 -1
  20. package/dist/llm/index.js +77 -5
  21. package/dist/mcp/index.js +0 -1
  22. package/dist/repl/index.js +21 -15
  23. package/dist/runtime/browser-manager.js +299 -0
  24. package/dist/runtime/dev-server-manager.js +354 -0
  25. package/dist/runtime/shutdown.js +26 -0
  26. package/dist/session/compact.js +86 -102
  27. package/dist/session/index.js +0 -1
  28. package/dist/session/scheduler.js +88 -92
  29. package/dist/session/trace-metrics.js +5 -92
  30. package/dist/session/trace.js +1 -10
  31. package/dist/tools/builtins/browser.js +199 -0
  32. package/dist/tools/builtins/dev-server.js +99 -0
  33. package/dist/tools/builtins/index.js +28 -19
  34. package/dist/tools/builtins/screenshot.js +173 -0
  35. package/dist/tools/builtins/view-image.js +49 -0
  36. package/dist/tools/constants.js +3 -0
  37. package/dist/tools/registry.js +5 -29
  38. package/dist/ui/layout.js +28 -5
  39. package/dist/ui/render.js +11 -0
  40. package/package.json +2 -2
  41. package/dist/agent/middleware/checklist.js +0 -59
  42. package/dist/session/drop.d.ts +0 -19
  43. package/dist/session/drop.js +0 -93
  44. package/dist/tools/builtins/drop-context.d.ts +0 -18
  45. package/dist/tools/builtins/drop-context.js +0 -68
  46. package/dist/verification/diagnostics.js +0 -108
  47. package/dist/verification/fingerprint.js +0 -54
  48. package/dist/verification/index.js +0 -333
  49. package/dist/verification/postconditions.js +0 -98
  50. package/dist/verification/targeted-tests.js +0 -96
  51. package/dist/verification/types.js +0 -1
@@ -15,7 +15,6 @@ import { createPetHooks } from '../pet/state.js';
15
15
  import { t } from '../i18n/index.js';
16
16
  import { isToolErrorOutput } from '../tools/result.js';
17
17
  import { appendCurrentSessionTraceEvent } from '../session/index.js';
18
- import { buildActiveNotesPlanReminder } from '../session/notes-plan.js';
19
18
  /** 当前 turn 的 batch id(runAgent 内闭包变量;一条 turn 一轮 tool batch 结束即清空)。 */
20
19
  let currentBatchId = null;
21
20
  let turnFileChanges = [];
@@ -202,6 +201,8 @@ onContextUpdate) {
202
201
  },
203
202
  onStepStart: () => spinner.start(t('agent.thinking')),
204
203
  onChatDone: () => spinner.stop(),
204
+ // 流式实时用量 → 底栏 context 进度条左侧 chip;轮末由 repl 清空。
205
+ onLiveUsage: (u) => layout.setLiveUsage(u),
205
206
  onTextEnd: () => {
206
207
  if (lastChar && lastChar !== '\n') {
207
208
  layout.contentWrite('\n');
@@ -240,24 +241,6 @@ onContextUpdate) {
240
241
  flushToolBatch();
241
242
  layout.contentWrite(`${ui.dim}${t('agent.aborted')}${ui.reset}\n`);
242
243
  },
243
- onValidationStart: (command) => {
244
- flushToolBatch();
245
- spinner.start(t('agent.validating', { command }));
246
- },
247
- onValidationResult: (validation) => {
248
- spinner.stop();
249
- const color = validation.status === 'passed'
250
- ? ui.green
251
- : validation.status === 'failed'
252
- ? ui.red
253
- : ui.yellow;
254
- const command = validation.command ?? t('agent.validationNoCommand');
255
- const detail = validation.status === 'skipped' && validation.skipReason
256
- ? `${validation.status}: ${validation.skipReason}`
257
- : validation.status;
258
- const symbol = validation.status === 'passed' ? '●' : validation.status === 'failed' ? '×' : '!';
259
- layout.contentWrite(` ${color}${symbol}${ui.reset} ${t('agent.validationResult', { command, status: detail })}\n\n`);
260
- },
261
244
  onDone: (elapsedMs, usage) => {
262
245
  flushToolBatch();
263
246
  writeChangeOverview();
@@ -281,9 +264,7 @@ onContextUpdate) {
281
264
  userInput,
282
265
  signal,
283
266
  onContextUpdate,
284
- dynamicSystemSuffix: buildActiveNotesPlanReminder,
285
267
  hooks: combinedHooks,
286
- autoValidate: config.autoValidate,
287
268
  onTraceEvent: appendCurrentSessionTraceEvent,
288
269
  });
289
270
  }
@@ -31,8 +31,8 @@ const SUBAGENT_ROLE = `## Sub-agent execution
31
31
  You are executing one delegated sub-task with the same engineering standards and capabilities as mocode.
32
32
  - Treat Task context as authoritative facts already established by the main agent; do not rediscover them without evidence they are stale.
33
33
  - Focus on the delegated scope, but continue until it is genuinely complete. Do not stop to save tokens.
34
- - Do not recursively call sub-agent. A write task runs in an isolated overlay; the coordinator merges and performs final unified verification.
35
- - Return concise findings, changes, verification evidence, and blockers to the coordinator.`;
34
+ - Do not recursively call sub-agent. A write task runs in an isolated overlay; the coordinator merges it safely.
35
+ - Return concise findings, changes, checks you chose to run, and blockers to the coordinator.`;
36
36
  /**
37
37
  * 派生一个子 agent 执行独立子任务。
38
38
  *
@@ -51,7 +51,7 @@ export async function spawnAgent(opts) {
51
51
  summary: null,
52
52
  completed: false,
53
53
  transcript: 'Sub-agent execution is disabled. Enable it with /subagent on.',
54
- status: 'failed', findings: [], readSet: [], changeSet: null, verification: null,
54
+ status: 'failed', findings: [], readSet: [], changeSet: null,
55
55
  usage: { promptTokens: 0, completionTokens: 0, totalTokens: 0, cachedTokens: 0, reasoningTokens: 0 },
56
56
  };
57
57
  }
@@ -135,7 +135,6 @@ export async function spawnAgent(opts) {
135
135
  maxSteps,
136
136
  toolsOverride,
137
137
  contextState: localContextState,
138
- autoValidate: false,
139
138
  onToolOutcome: (tool, args) => {
140
139
  if (tool === 'read_file' && typeof args.path === 'string')
141
140
  readSet.add(args.path);
@@ -173,7 +172,6 @@ export async function spawnAgent(opts) {
173
172
  findings: result.finalText ? [result.finalText] : [],
174
173
  readSet: [...readSet].sort(),
175
174
  changeSet,
176
- verification: null, // 主 Agent 在所有 coordinator merge 完成后统一验证
177
175
  usage: {
178
176
  promptTokens: result.usage?.promptTokens ?? 0,
179
177
  completionTokens: result.usage?.completionTokens ?? 0,
@@ -1,18 +1,5 @@
1
- // PROMPT-01: Build-and-Self-Verify working discipline.
2
- //
3
- // 把"完成必须验证"作为 system prompt 的一等公民,而不是事后选项。注入一段
4
- // 4 阶段纪律(Plan & Discover → Build → Verify → Fix),并提供 per-model
5
- // 措辞(anthropic / openai / qwen)让 base model 拿到最适合自己的表述。
6
- //
7
- // 关键约束:
8
- // - 纯函数,无副作用,无 config 依赖 → 不踩 TDZ,易测,易回滚。
9
- // - 段标题在 buildMocodeCorePrompt 之外,不会被 `## Project context` 索引
10
- // 切片误伤;且 buildBasePrompt 注入位置在 ## Workflow 之前,确保 LLM
11
- // 先看到纪律再看工具/平台细节。
12
- // - per-model 措辞是"轻量"差异:3 个家族共享 4 阶段结构,只在首句
13
- // 上贴近该家族的指令遵从习惯;真正的 prompt 反演化交给 AHE。
14
- // - 语种统一英文:4 份都用同一份核心纪律文本,避免多语种漂移;用户语言
15
- // 偏好由现有 i18n 段(assistant.languageInstruction)负责。
1
+ // Lightweight, advisory working guidance. The agent decides how much discovery and
2
+ // validation each task needs; the framework does not enforce a completion gate.
16
3
  /**
17
4
  * 从 config.model 字符串里嗅探 model family。匹配规则尽量宽松,够用即可。
18
5
  * 未来 AHE 闭环后可以换成 config.modelFamily 字段。
@@ -33,73 +20,32 @@ export function inferModelFamily(model) {
33
20
  * 4 阶段核心纪律(英文)。4 个 model family 共用此文本,只在首句与标题
34
21
  * 标签上做轻量变体。保持短小,详细的完成检查由动态 checklist 按需注入。
35
22
  */
36
- const CORE_SECTION = `## Working discipline — coding tasks (Build-and-Self-Verify)
37
-
38
- Treat "verification" as a first-class part of the task, not an afterthought. Use the smallest evidence-driven loop below.
39
-
40
- ### Phase 1 — Plan & Discover
41
- - Open with a one-sentence restatement of your interpretation of the request; if a materially different reading exists, name it briefly before proceeding. This catches misunderstanding before any work is wasted.
42
- - State the goal and a concrete acceptance signal, then inspect the relevant code before changing it.
43
- - Ask only when an unresolved choice is high-impact or user-owned; otherwise follow repository evidence and proceed.
44
-
45
- ### Phase 2 — Build
46
- - Make the smallest coherent change; avoid unrelated refactors.
47
- - Add or update a focused test when behavior changes and the project has an applicable test suite.
48
- - Re-read only when a dependent edit needs fresh exact content or state may be stale.
49
-
50
- ### Phase 3 — Verify
51
- - Run the smallest executable check that proves the requested behavior, then read its complete result.
52
- - Compare evidence with the user's request, not merely with the diff.
53
-
54
- ### Phase 4 — Fix
55
- - Diagnose the root cause, make a focused correction, and rerun the relevant check.
56
- - After two identical failures, change the approach instead of repeating the same call.
57
-
58
- **Hard rule (non-negotiable):** "I read the code and it looks right" is not a completion signal. Report the verification performed, or state clearly why it could not be run.
59
-
60
- **Hard rule (non-negotiable):** Never invent file paths, APIs, config keys, flags, or behavior. Every claim about the codebase must trace to tool output in this conversation; explicitly label anything you have not verified as an assumption.`;
61
- /**
62
- * 把核心段适配到指定 model family:只替换首行(语序 / 强动词),段标题
63
- * 保持原样。Phase 内容保持原样,4 份共享同一份结构化文本。
64
- * 注意:不再往标题注入 "[model: X]" 标签——它对模型是无意义噪声,
65
- * 还可能引发自我指涉,反而干扰遵从。
66
- */
67
- function adapt(_model, opener) {
68
- return CORE_SECTION.replace('Treat "verification" as a first-class part of the task, not an afterthought.', opener);
69
- }
23
+ const CORE_SECTION = `## Working discipline — coding tasks
24
+
25
+ Use your judgment to choose the shortest reliable path from the request to a useful result.
26
+
27
+ - Inspect only the code and context needed for the next decision.
28
+ - Make the smallest coherent change and avoid unrelated refactors.
29
+ - Decide whether validation is useful based on risk, scope, available commands, and the user's request. Validation is optional, not a completion gate.
30
+ - When validation is useful, choose the smallest relevant check yourself; do not run broad test/build suites by default.
31
+ - Re-read or rerun only when evidence is stale or the next edit depends on exact current content.
32
+ - On failure, diagnose before retrying; after repeated identical failures, change approach.
33
+ - Report honestly what you changed, what you checked, and anything left uncertain.
34
+
35
+ Never invent file paths, APIs, config keys, flags, or behavior. Distinguish repository evidence from assumptions.`;
70
36
  /** ASK-01: only user-owned, high-impact choices should interrupt autonomous execution. */
71
- const ASK_WHITELIST_SECTION = `## When to ask instead of guess
72
-
73
- Call \`ask_human\` before coding only when repository evidence cannot resolve a user-owned, high-impact choice:
74
- 1. irreversible deletion, migration, security, permission, or external side effect;
75
- 2. public API compatibility (keep, deprecate, rename, or remove);
76
- 3. multiple reasonable options that materially change product behavior;
77
- 4. the request itself admits two or more materially different readings that lead to different deliverables (do not silently pick one and guess).
78
-
79
- For naming, implementation detail, and verification commands, follow repository precedent and choose the safest reversible default. Disclose any consequential assumption.
80
-
37
+ const ASK_WHITELIST_SECTION = `## When to ask instead of guess
38
+
39
+ Call \`ask_human\` before coding only when repository evidence cannot resolve a user-owned, high-impact choice:
40
+ 1. irreversible deletion, migration, security, permission, or external side effect;
41
+ 2. public API compatibility (keep, deprecate, rename, or remove);
42
+ 3. multiple reasonable options that materially change product behavior;
43
+ 4. the request itself admits two or more materially different readings that lead to different deliverables (do not silently pick one and guess).
44
+
45
+ For naming and implementation details, follow repository precedent and choose the safest reversible default. Disclose any consequential assumption.
46
+
81
47
  Budget: at most 2 \`ask_human\` calls per turn. Beyond that, use the safest reversible default and disclose it in the final reply.`;
82
- /**
83
- * 拼出纪律段 + ASK-01 卡点白名单。返回完整段(两段用 \`\\n\\n\` 隔开);
84
- * 工厂之前只返回纪律段,ASK-01 落地后变成纪律 + 白名单两段;
85
- * ASK-01 段是固定英文,不参与 per-model 适配(避免 4 份变体维护成本)。
86
- */
87
- export function buildWorkDisciplineSection(modelFamily) {
88
- let section;
89
- switch (modelFamily) {
90
- case 'anthropic':
91
- section = adapt('anthropic', 'Verification is a hard prerequisite for completion, not a courtesy.');
92
- break;
93
- case 'openai':
94
- section = adapt('openai', 'Every coding task MUST complete these four phases in order. Skipping or merging phases is treated as a failure. For trivial or read-only requests, phases may collapse.');
95
- break;
96
- case 'qwen':
97
- section = adapt('qwen', 'Verification is a hard prerequisite for completion; "I wrote the code" is not evidence the code works.');
98
- break;
99
- case 'other':
100
- case undefined:
101
- default:
102
- section = CORE_SECTION;
103
- }
104
- return `${section}\n\n${ASK_WHITELIST_SECTION}`;
48
+ /** Advisory guidance shared by main and sub-agents. */
49
+ export function buildWorkDisciplineSection(_modelFamily) {
50
+ return `${CORE_SECTION}\n\n${ASK_WHITELIST_SECTION}`;
105
51
  }
@@ -41,6 +41,7 @@ export const languageFromShell = process.env.MOCODE_LANGUAGE !== undefined;
41
41
  // 仿 themeFromShell 模式:shell export 的环境变量在 loadEnvFiles 中不被回填(优先级最高),
42
42
  // 故 /model 写入 ~/.mocode/config 的同名键下次启动会被 shell 值覆盖——据此给 dim 警告。
43
43
  const LLM_ENV_KEYS = ['LLM_BASE_URL', 'LLM_API_KEY', 'LLM_MODEL', 'CONTEXT_WINDOW_TOKENS'];
44
+ export const DEFAULT_CONTEXT_WINDOW_TOKENS = 256000;
44
45
  const llmKeysFromShell = LLM_ENV_KEYS.filter((k) => process.env[k] !== undefined);
45
46
  loadEnvFiles();
46
47
  setLanguage(detectLanguage(process.env.MOCODE_LANGUAGE));
@@ -195,11 +196,12 @@ ${buildWorkDisciplineSection(inferModelFamily(config.model))}
195
196
 
196
197
  ## Workflow
197
198
  - Use existing conversation and tool evidence before gathering more. Inspect only what supports the next decision; do not guess.
198
- - Keep changes focused. After modifications, run the smallest relevant executable verification that actually exercises the requested behavior, and report its result. A command exiting 0 is not proof the task is done — confirm the specific behavior the user asked for is observed, not merely that the diff applied.
199
+ - Keep changes focused. Decide for yourself whether a check is worth running; prefer the smallest relevant check and avoid broad test/build suites unless the task or risk justifies them.
199
200
  - Use web search only when freshness materially affects the answer.
200
201
  ${buildCodegraphSection()}
201
202
 
202
203
  ## Tool use
204
+ - During tool-calling turns, stay silent unless something important enough must reach the user — otherwise just call the tool and let it run.
203
205
  - Go directly to a known path or symbol; use discovery tools only when the location is unknown.
204
206
  - Edit against a FRESH read: before any edit_file/write_file, call read_file on the exact path and copy both its latest hash and the exact target text. Never reconstruct old_string from a grep/summary/diff — those lose whitespace and indentation and cause edit failures.
205
207
  - A read_file hash from before a compaction, session resume, edit conflict, or external change is STALE and will be rejected — re-read rather than reuse an old hash.
@@ -297,16 +299,14 @@ export const config = {
297
299
  get systemPrompt() {
298
300
  return buildBasePrompt();
299
301
  },
300
- contextWindowTokens: Number(process.env.CONTEXT_WINDOW_TOKENS) || 128000,
301
- compactThreshold: Number(process.env.COMPACT_THRESHOLD) || 0.85,
302
+ contextWindowTokens: Number(process.env.CONTEXT_WINDOW_TOKENS) || DEFAULT_CONTEXT_WINDOW_TOKENS,
302
303
  includeUsage: process.env.LLM_STREAM_USAGE !== 'false',
303
304
  autoCompact: process.env.AUTO_COMPACT !== 'false',
304
- autoValidate: process.env.MOCODE_AUTO_VALIDATE !== 'false',
305
- contextOptimize: process.env.MOCODE_CONTEXT_OPTIMIZE !== 'false',
306
- contextRelprune: process.env.MOCODE_CONTEXT_RELPRUNE !== 'false',
305
+ contextOptimize: process.env.MOCODE_CONTEXT_OPTIMIZE === 'true',
306
+ contextRelprune: process.env.MOCODE_CONTEXT_RELPRUNE === 'true',
307
307
  contextLifecycle: process.env.MOCODE_LIFECYCLE !== 'false',
308
308
  contextBudget: process.env.MOCODE_BUDGET_SCHEDULER !== 'false',
309
- autoReflect: process.env.AUTO_REFLECT !== 'false',
309
+ autoReflect: process.env.AUTO_REFLECT === 'true',
310
310
  memoryEnabled: process.env.MEMORY_ENABLED === 'true',
311
311
  reflectEveryN: Number(process.env.REFLECT_EVERY_N) || 5,
312
312
  maxSteps: Number(process.env.MAX_STEPS) || 1000,
@@ -1,12 +1,8 @@
1
- // Age-aware tool-result encoding coordinator.
2
- // Initial pushes stay conservative; old Cold results are re-encoded before chat.
3
- import { TOOL_OLD_AGE } from './budget.js';
1
+ // Pressure-only tool-result encoding coordinator.
2
+ // Normal pushes are raw apart from the per-result hard safety cap.
4
3
  import { optimizeToolResult } from './pipeline.js';
5
4
  import { canonicalizePath, extractPath, isToolResultSuccess, toText, } from './utils.js';
6
- /**
7
- * Tracks successful first reads and tool-result age without coupling encoders to
8
- * lifecycle's mutable history indexes. All methods are fail-safe and idempotent.
9
- */
5
+ /** Rebuilds records from current history for each pressure pass. */
10
6
  export class AgeAwareEncodingState {
11
7
  pushOrdinal = 0;
12
8
  records = new Map();
@@ -14,36 +10,16 @@ export class AgeAwareEncodingState {
14
10
  constructor(history = []) {
15
11
  this.rehydrate(history);
16
12
  }
17
- /** Build the conservative context for a newly completed tool result. */
18
- preparePush(tc, succeeded) {
19
- const path = tc.name === 'read_file'
20
- ? canonicalizePath(extractPath(tc.arguments))
21
- : null;
22
- const isFirstRead = path ? !this.seenReadPaths.has(path) : undefined;
23
- this.records.set(tc.id, {
24
- toolCallId: tc.id,
25
- toolName: tc.name,
26
- argsRaw: tc.arguments,
27
- pushOrdinal: this.pushOrdinal,
28
- succeeded,
29
- isFirstRead,
30
- agedEncoded: false,
31
- });
32
- this.pushOrdinal++;
33
- // Failed reads must not consume the "first successful read" privilege.
34
- if (succeeded && path)
35
- this.seenReadPaths.add(path);
36
- return {
37
- age: 0,
38
- isCold: false,
39
- isFirstRead,
40
- phase: 'push',
41
- };
42
- }
43
- /** Re-encode eligible tool messages in the Cold prefix in place. */
44
- sweep(history, hotBoundary) {
13
+ /**
14
+ * Pressure-only, progressive encoding of Cold logs and retrievable searches.
15
+ * It never touches code reads, skills, human decisions, or sub-agent output;
16
+ * repeated pressure passes may further reduce content only when strictly shorter.
17
+ */
18
+ sweepPressure(history, hotBoundary) {
45
19
  try {
20
+ const pressureEncodable = new Set(['run_command', 'grep', 'glob', 'web_search', 'web_fetch']);
46
21
  const end = Math.min(Math.max(hotBoundary, 1), history.length);
22
+ let encodedCount = 0;
47
23
  for (let idx = 1; idx < end; idx++) {
48
24
  const message = history[idx];
49
25
  if (message.role !== 'tool')
@@ -51,32 +27,27 @@ export class AgeAwareEncodingState {
51
27
  const toolMessage = message;
52
28
  const id = toolMessage.tool_call_id;
53
29
  const record = id ? this.records.get(id) : undefined;
54
- if (!record || !record.succeeded || record.agedEncoded)
30
+ if (!record || !record.succeeded || !pressureEncodable.has(record.toolName))
55
31
  continue;
56
32
  const content = toText(toolMessage.content);
57
- if (!content || content.startsWith('⌦[')) {
58
- record.agedEncoded = true;
33
+ if (!content || content.startsWith('⌦['))
59
34
  continue;
60
- }
61
- // Exclude the result's own push: immediately after insertion its age is 0.
62
35
  const age = Math.max(0, this.pushOrdinal - record.pushOrdinal - 1);
63
- if (age < TOOL_OLD_AGE)
64
- continue;
65
36
  const encoded = optimizeToolResult(record.toolName, content, record.argsRaw, {
66
37
  age,
67
38
  isCold: true,
68
39
  isFirstRead: record.isFirstRead,
69
40
  phase: 'sweep',
70
41
  });
71
- // Aged encoding is a degradation step: never replace content with a
72
- // representation that is equal-sized or larger.
73
- if (encoded.length < content.length)
42
+ if (encoded.length < content.length) {
74
43
  toolMessage.content = encoded;
75
- record.agedEncoded = true;
44
+ encodedCount++;
45
+ }
76
46
  }
47
+ return encodedCount;
77
48
  }
78
49
  catch {
79
- // Context optimization must never block an agent request.
50
+ return 0;
80
51
  }
81
52
  }
82
53
  /** Rebuild stable state after resume or structural history compaction. */
@@ -119,7 +90,6 @@ export class AgeAwareEncodingState {
119
90
  pushOrdinal: this.pushOrdinal,
120
91
  succeeded,
121
92
  isFirstRead,
122
- agedEncoded: false,
123
93
  });
124
94
  this.pushOrdinal++;
125
95
  if (succeeded && path)
@@ -132,10 +132,17 @@ export function recordArtifact(state, history, idx, output, succeeded) {
132
132
  updateStats(state, stateFor(state));
133
133
  }
134
134
  function affected(artifact, changed) {
135
- return artifact.dependencies.some((dependency) => dependency.path === '*' || changed.has(dependency.path));
135
+ // '*' 依赖(无法解析出具体文件路径的诊断/搜索结果)不与任何具体写操作关联:
136
+ // 任何文件写入都会作废全部 '*' artifact,等于每次 mutation 都销毁
137
+ // git/测试/构建等历史证据,模型被迫反复 re-run,轮次爆炸。只失效路径明确命中的。
138
+ return artifact.dependencies.some((dependency) => dependency.path !== '*' && changed.has(dependency.path));
136
139
  }
137
- /** Mark and immediately stub stale facts; this is stronger than waiting for budget pressure. */
138
- export function invalidateArtifacts(state, history, changedFiles) {
140
+ /**
141
+ * Mark precise, file-backed facts stale without changing the evidence in history.
142
+ * A stale result can still explain a later edit or failure; pressure compression is
143
+ * the only path allowed to replace its content with a compact marker.
144
+ */
145
+ export function invalidateArtifacts(state, _history, changedFiles) {
139
146
  const changed = new Set(changedFiles.map(canonicalizePath).filter((item) => !!item));
140
147
  if (changed.size === 0)
141
148
  return 0;
@@ -145,15 +152,6 @@ export function invalidateArtifacts(state, history, changedFiles) {
145
152
  if (artifact.freshness !== 'fresh' || !affected(artifact, changed))
146
153
  continue;
147
154
  artifact.freshness = 'stale';
148
- const message = history[artifact.messageIndex];
149
- if (message?.role === 'tool') {
150
- const original = toText(message.content);
151
- const paths = artifact.dependencies.map((item) => item.path).join(', ');
152
- const stub = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id} dependencies=${paths} ` +
153
- `invalidated-by=${[...changed].join(', ')}; re-run ${artifact.source.tool} before using this fact.`;
154
- message.content = stub;
155
- artifact.tokenCount = estimateTokens(stub);
156
- }
157
155
  count++;
158
156
  }
159
157
  updateStats(state, artifactState);
@@ -193,7 +191,7 @@ export function rehydrateArtifacts(state, history) {
193
191
  }
194
192
  updateStats(state, artifactState);
195
193
  }
196
- /** Compare captured dependency versions before each model step to detect external edits. */
194
+ /** Compare captured dependency versions before each model step and mark stale metadata only. */
197
195
  export function refreshArtifactFreshness(state, history) {
198
196
  const changed = new Set();
199
197
  for (const artifact of stateFor(state).artifacts.values()) {
@@ -208,19 +206,23 @@ export function refreshArtifactFreshness(state, history) {
208
206
  }
209
207
  return changed.size > 0 ? invalidateArtifacts(state, history, [...changed]) : 0;
210
208
  }
211
- /** Scheduler entry point: stale artifacts are already stubs; normalize any resumed stale message first. */
212
- export function pruneStaleArtifacts(state, history) {
209
+ /**
210
+ * Pressure-only stage: replace stale evidence in the Cold prefix with a compact
211
+ * marker. Recent/current work stays intact, and normal mutation handling never
212
+ * calls this function.
213
+ */
214
+ export function pruneStaleArtifacts(state, history, coldBoundary) {
213
215
  const artifactState = stateFor(state);
214
216
  let pruned = 0;
215
217
  for (const artifact of artifactState.artifacts.values()) {
216
- if (artifact.freshness !== 'stale')
218
+ if (artifact.freshness !== 'stale' || artifact.messageIndex >= coldBoundary)
217
219
  continue;
218
220
  const message = history[artifact.messageIndex];
219
221
  if (message?.role !== 'tool')
220
222
  continue;
221
223
  const content = toText(message.content);
222
224
  if (!content.startsWith(STALE_PREFIX)) {
223
- message.content = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id}; re-run before use.`;
225
+ message.content = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id}; source changed, so this content may be outdated.`;
224
226
  artifact.tokenCount = estimateTokens(String(message.content));
225
227
  pruned++;
226
228
  }
@@ -1,12 +1,8 @@
1
- // 五区 Context Budget Scheduler。
1
+ // 五区 Context Budget accounting and the shared pressure threshold.
2
2
  //
3
- // 目的:把当前请求拆成 System / History / Tool-Recent / Tool-Old / Summary + Reserve,
4
- // 统一报告各区占用,并只调度执行层能够真正落地的 warn / compact_history。
5
- //
6
- // push-time cap、pipeline、relevance、lifecycle 与 age-aware sweep 负责工具结果优化;
7
- // scheduler 在这些处理完成后评估,不重复生成 Cold/Hot tool 压缩动作。
8
- // 本文件保持叶子级,只依赖 ChatMessage / token estimate,具体执行由 session/scheduler.ts 完成。
9
- // contextBudget 开关关闭时,agent/core.ts 退化为直接调用 maybeCompact。
3
+ // This module only estimates and reports. session/scheduler.ts owns the sole
4
+ // automatic rewrite sequence and starts all pressure cleanup plus compact_history
5
+ // when corrected or raw request occupancy reaches 80%.
10
6
  import { chatTools, estimateMessagesTokens, estimateToolSchemaTokens, messageTokens, } from '../llm/index.js';
11
7
  import { toText } from './utils.js';
12
8
  /** 五区分账(占比对齐 CONTEXT_WINDOW)。顺序固定,便于遍历。 */
@@ -28,9 +24,8 @@ export const DEFAULT_BUDGET_POLICY = {
28
24
  reserve: 0.05,
29
25
  },
30
26
  hotTurnWindow: 4,
31
- toolOldAge: 2,
32
27
  compactKeepRatio: 0.40,
33
- totalTriggerRatio: 0.82,
28
+ pressureTriggerRatio: 0.80,
34
29
  schedulerTargetRatio: 0.80,
35
30
  estimateSafetyFactor: 1.05,
36
31
  compactHeadroomTokens: 1500,
@@ -38,7 +33,6 @@ export const DEFAULT_BUDGET_POLICY = {
38
33
  /** 兼容既有调用方的只读别名;配置只在 DEFAULT_BUDGET_POLICY 中维护。 */
39
34
  export const BUDGET_RATIO = DEFAULT_BUDGET_POLICY.ratios;
40
35
  export const HOT_TURN_WINDOW = DEFAULT_BUDGET_POLICY.hotTurnWindow;
41
- export const TOOL_OLD_AGE = DEFAULT_BUDGET_POLICY.toolOldAge;
42
36
  function msgTokens(m) {
43
37
  // 与请求预估复用同一实现,避免角色结构开销、多模态和 tool_calls 在两个预算路径中漂移。
44
38
  return messageTokens(m);
@@ -69,13 +63,17 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
69
63
  }
70
64
  // 校正后的 token 数:raw * correction,最小 1(raw > 0 时)。
71
65
  const adj = (raw) => (raw > 0 ? Math.max(1, Math.round(raw * correction)) : 0);
66
+ // 裸总量(不乘 correction):硬闸用它判断,防止 correction 折扣否决真实溢出。
67
+ let rawTotal = 0;
72
68
  const sysMsg = history[0];
73
69
  const systemCosts = {
74
70
  prompt: sysMsg ? msgTokens(sysMsg) : 0,
75
71
  toolSchemas: estimateToolSchemaTokens(activeTools),
76
72
  };
77
73
  // 工具 schema 与 system prompt 同属请求固定开销;必须计入总量才能可靠触发压缩。
78
- layers.system.actual = adj(systemCosts.prompt + systemCosts.toolSchemas);
74
+ const systemRaw = systemCosts.prompt + systemCosts.toolSchemas;
75
+ layers.system.actual = adj(systemRaw);
76
+ rawTotal += systemRaw;
79
77
  // Summary 检测:role:'system' 且不是 history[0] 的,视为摘要(compact.ts 摘要插 index 1)。
80
78
  // 简单启发:若 history[1]?.role === 'system' 且 content 含「# 会话摘要」特征串,计入 summary。
81
79
  // 命中时循环跳过 i=1;不命中时当作普通 message(罕见,落到下方 user/assistant 分支)。
@@ -83,7 +81,9 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
83
81
  if (history.length > 1 && history[1].role === 'system') {
84
82
  const c1 = toText(history[1].content);
85
83
  if (c1.startsWith('# 会话摘要') || c1.includes('会话摘要')) {
86
- layers.summary.actual = adj(msgTokens(history[1]));
84
+ const summaryRaw = msgTokens(history[1]);
85
+ layers.summary.actual = adj(summaryRaw);
86
+ rawTotal += summaryRaw;
87
87
  summaryHit = true;
88
88
  }
89
89
  }
@@ -94,16 +94,18 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
94
94
  const m = history[i];
95
95
  if (i === 1 && summaryHit)
96
96
  continue; // summary 已单独算过
97
+ const raw = msgTokens(m);
97
98
  if (m.role === 'tool') {
98
- const t = adj(msgTokens(m));
99
99
  if (i >= hotStart)
100
- layers.toolRecent.actual += t;
100
+ layers.toolRecent.actual += adj(raw);
101
101
  else
102
- layers.toolOld.actual += t;
102
+ layers.toolOld.actual += adj(raw);
103
+ rawTotal += raw;
103
104
  }
104
105
  else if (m.role !== 'system') {
105
106
  // user / assistant 全部计入 history(对话轨迹)
106
- layers.history.actual += adj(msgTokens(m));
107
+ layers.history.actual += adj(raw);
108
+ rawTotal += raw;
107
109
  }
108
110
  // 其它 system(几乎不存在)跳过
109
111
  }
@@ -120,10 +122,11 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
120
122
  // 按 overRatio 降序
121
123
  triggers.sort((a, b) => layers[b].overRatio - layers[a].overRatio);
122
124
  const total = BUDGET_LAYERS.reduce((s, k) => s + (k === 'reserve' ? 0 : layers[k].actual), 0);
123
- const totalOver = total >= DEFAULT_BUDGET_POLICY.totalTriggerRatio * window;
125
+ const totalOver = total >= DEFAULT_BUDGET_POLICY.pressureTriggerRatio * window;
124
126
  return {
125
127
  step,
126
128
  total,
129
+ rawTotal,
127
130
  window,
128
131
  layers,
129
132
  systemCosts,
@@ -134,15 +137,11 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
134
137
  };
135
138
  }
136
139
  /** 根据 BudgetReport 生成可执行动作。
137
- * push-time cap、relevance、lifecycle 与 age-aware sweep 已在评估前完成,
138
- * 因此这里不再生成无法执行的 Cold/Hot tool action。
139
- * History 或总量超预算时才考虑昂贵的 LLM 摘要。 */
140
+ * History compaction is the final fallback after pressure-only tool stages.
141
+ * Per-layer overages are diagnostics, never independent rewrite triggers. */
140
142
  export function scheduleActions(report) {
141
143
  const actions = [];
142
- const { layers, totalOver, total } = report;
143
- const policy = DEFAULT_BUDGET_POLICY;
144
- const headroom = policy.schedulerTargetRatio * report.window
145
- - total * policy.estimateSafetyFactor;
144
+ const { layers } = report;
146
145
  if (layers.system.overBudget) {
147
146
  const { prompt, toolSchemas } = report.systemCosts;
148
147
  const { actual, budget } = layers.system;
@@ -155,8 +154,8 @@ export function scheduleActions(report) {
155
154
  + `提示 ${prompt} + 工具 ${toolSchemas},×${report.correction.toFixed(2)}。`,
156
155
  });
157
156
  }
158
- if ((layers.history.overBudget || totalOver)
159
- && headroom < -policy.compactHeadroomTokens) {
157
+ const pressureLine = DEFAULT_BUDGET_POLICY.pressureTriggerRatio * report.window;
158
+ if (Math.max(report.rawTotal, report.total) >= pressureLine) {
160
159
  actions.push({ kind: 'compact_history' });
161
160
  }
162
161
  return actions;
@@ -164,7 +163,7 @@ export function scheduleActions(report) {
164
163
  /** 拍平成人类可读(供 /context 命令与 check-budget 脚本用)。 */
165
164
  export function formatReport(report) {
166
165
  const lines = [];
167
- lines.push(`step ${report.step} total ${report.total}/${report.window} (${((report.total / report.window) * 100).toFixed(1)}%)`);
166
+ lines.push(`step ${report.step} total ${report.total}/${report.window} (${((report.total / report.window) * 100).toFixed(1)}%) raw ${report.rawTotal}`);
168
167
  for (const k of BUDGET_LAYERS) {
169
168
  const lb = report.layers[k];
170
169
  const pct = lb.budget > 0 ? ((lb.actual / lb.budget) * 100).toFixed(0) : '-';
@@ -29,7 +29,6 @@ const BY_NAME = {
29
29
  edit_file: 'status',
30
30
  write_file: 'status',
31
31
  ask_human: 'status',
32
- drop_context: 'status',
33
32
  memory_save: 'status',
34
33
  memory_update: 'status',
35
34
  memory_forget: 'status',
@@ -1,14 +1,7 @@
1
- // 内置 encoder 清单。启动期 pipeline 首次调用时经 registerAll 注册到 registry。
2
- //
3
- // Phase 1:passthrough(identity 兜底)→ 全链路零行为变化。
4
- // Phase 2:tree / search / log / table / memory(高价值低风险)。
5
- // Phase 2.5:code / graph / doc / summary(覆盖剩余有 encoder 的 kind)。
6
- // - code(read_file):仅折叠 ≥3 连续空行,行号保真(edit_file 依赖)。
7
- // - graph(保留 kind,暂无 builtin 工具直接命中)/ doc(web_fetch, use_skill)/ summary(task):
8
- // 去 ANSI + 折叠空行,保守不重构结构。
9
- // - status(edit/write/ask_human/mem 增删改)无 encoder:本就是一行,无需编码。
10
- //
11
- // 加 encoder:新建 encoders/xxx.ts 导出 ContextEncoder,在此数组加一行。无需动 agent / llm / core。
1
+ // 内置 encoder 清单。仅由 opt-in pressure stage 首次调用时懒注册。
2
+ // Normal tool pushes do not pass through these encoders.
3
+ // Available transforms:tree / search / log / table / memory / code / doc / summary.
4
+ // status 类工具本就是短结果,无专用 encoder。
12
5
  import { passthroughEncoder } from './passthrough.js';
13
6
  import { treeEncoder } from './tree.js';
14
7
  import { searchEncoder } from './search.js';
@@ -1,12 +1,9 @@
1
- // context/ barrel:Context Optimization Pipeline + 五区 Budget Scheduler。
2
- //
3
- // 单一入口 optimizeToolResult(agent/core.ts pushToolResult 调)接管"工具结果进 LLM 前"的表示。
4
- // 单一入口 runScheduler(agent/core.ts 步前调)接管"何时调用哪一闸"的调度。
5
- // 不调 LLM、不碰 Tool Calling schema / executeTool / tool_call_id 配对 / TUI 渲染
6
- // (叶子级:仅 stdlib + tools/constants + session/compact 的 capToolResultForHistory 兜底 + config 开关)。
1
+ // context/ barrel: metadata tracking, optional pressure encoders, and budget reporting.
2
+ // Normal tool pushes stay raw apart from the hard per-result cap. The session
3
+ // scheduler is the only automatic rewrite coordinator at real pressure.
7
4
  export { optimizeToolResult } from './pipeline.js';
8
5
  export { classify, knownToolKinds } from './classifier.js';
9
6
  export { recordArtifact, invalidateArtifacts, rehydrateArtifacts, refreshArtifactFreshness, pruneStaleArtifacts, collectArtifactRefs, formatArtifactTokenSources, } from './artifacts.js';
10
7
  export { registerEncoder, registerAll, getEncoder, registeredKinds, } from './registry.js';
11
8
  // ── Context Budget Scheduler ───────────────────────────────────────────────
12
- export { evaluateBudget, scheduleActions, formatReport, quickEstimate, userTurnBoundary, BUDGET_LAYERS, DEFAULT_BUDGET_POLICY, BUDGET_RATIO, HOT_TURN_WINDOW, TOOL_OLD_AGE, } from './budget.js';
9
+ export { evaluateBudget, scheduleActions, formatReport, quickEstimate, userTurnBoundary, BUDGET_LAYERS, DEFAULT_BUDGET_POLICY, BUDGET_RATIO, HOT_TURN_WINDOW, } from './budget.js';