mocode-ai 1.1.6 → 1.1.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +23 -37
- package/README.zh-CN.md +46 -38
- package/dist/agent/core.js +106 -395
- package/dist/agent/index.js +2 -21
- package/dist/agent/spawn.js +3 -5
- package/dist/agent/work-discipline.js +28 -82
- package/dist/config/index.js +7 -7
- package/dist/context/age-aware.js +18 -48
- package/dist/context/artifacts.js +19 -17
- package/dist/context/budget.js +27 -28
- package/dist/context/classifier.js +0 -1
- package/dist/context/encoders/index.js +4 -11
- package/dist/context/index.js +4 -7
- package/dist/context/lifecycle.js +115 -483
- package/dist/context/pipeline.js +8 -15
- package/dist/context/relevance.js +77 -55
- package/dist/host/stdio.js +0 -6
- package/dist/i18n/index.js +0 -6
- package/dist/index.js +11 -1
- package/dist/llm/index.js +77 -5
- package/dist/mcp/index.js +0 -1
- package/dist/repl/index.js +21 -15
- package/dist/runtime/browser-manager.js +299 -0
- package/dist/runtime/dev-server-manager.js +354 -0
- package/dist/runtime/shutdown.js +26 -0
- package/dist/session/compact.js +86 -102
- package/dist/session/index.js +0 -1
- package/dist/session/scheduler.js +88 -92
- package/dist/session/trace-metrics.js +5 -92
- package/dist/session/trace.js +1 -10
- package/dist/tools/builtins/browser.js +199 -0
- package/dist/tools/builtins/dev-server.js +99 -0
- package/dist/tools/builtins/index.js +28 -19
- package/dist/tools/builtins/screenshot.js +173 -0
- package/dist/tools/builtins/view-image.js +49 -0
- package/dist/tools/constants.js +3 -0
- package/dist/tools/registry.js +5 -29
- package/dist/ui/layout.js +28 -5
- package/dist/ui/render.js +11 -0
- package/package.json +2 -2
- package/dist/agent/middleware/checklist.js +0 -59
- package/dist/session/drop.d.ts +0 -19
- package/dist/session/drop.js +0 -93
- package/dist/tools/builtins/drop-context.d.ts +0 -18
- package/dist/tools/builtins/drop-context.js +0 -68
- package/dist/verification/diagnostics.js +0 -108
- package/dist/verification/fingerprint.js +0 -54
- package/dist/verification/index.js +0 -333
- package/dist/verification/postconditions.js +0 -98
- package/dist/verification/targeted-tests.js +0 -96
- package/dist/verification/types.js +0 -1
package/dist/agent/index.js
CHANGED
|
@@ -15,7 +15,6 @@ import { createPetHooks } from '../pet/state.js';
|
|
|
15
15
|
import { t } from '../i18n/index.js';
|
|
16
16
|
import { isToolErrorOutput } from '../tools/result.js';
|
|
17
17
|
import { appendCurrentSessionTraceEvent } from '../session/index.js';
|
|
18
|
-
import { buildActiveNotesPlanReminder } from '../session/notes-plan.js';
|
|
19
18
|
/** 当前 turn 的 batch id(runAgent 内闭包变量;一条 turn 一轮 tool batch 结束即清空)。 */
|
|
20
19
|
let currentBatchId = null;
|
|
21
20
|
let turnFileChanges = [];
|
|
@@ -202,6 +201,8 @@ onContextUpdate) {
|
|
|
202
201
|
},
|
|
203
202
|
onStepStart: () => spinner.start(t('agent.thinking')),
|
|
204
203
|
onChatDone: () => spinner.stop(),
|
|
204
|
+
// 流式实时用量 → 底栏 context 进度条左侧 chip;轮末由 repl 清空。
|
|
205
|
+
onLiveUsage: (u) => layout.setLiveUsage(u),
|
|
205
206
|
onTextEnd: () => {
|
|
206
207
|
if (lastChar && lastChar !== '\n') {
|
|
207
208
|
layout.contentWrite('\n');
|
|
@@ -240,24 +241,6 @@ onContextUpdate) {
|
|
|
240
241
|
flushToolBatch();
|
|
241
242
|
layout.contentWrite(`${ui.dim}${t('agent.aborted')}${ui.reset}\n`);
|
|
242
243
|
},
|
|
243
|
-
onValidationStart: (command) => {
|
|
244
|
-
flushToolBatch();
|
|
245
|
-
spinner.start(t('agent.validating', { command }));
|
|
246
|
-
},
|
|
247
|
-
onValidationResult: (validation) => {
|
|
248
|
-
spinner.stop();
|
|
249
|
-
const color = validation.status === 'passed'
|
|
250
|
-
? ui.green
|
|
251
|
-
: validation.status === 'failed'
|
|
252
|
-
? ui.red
|
|
253
|
-
: ui.yellow;
|
|
254
|
-
const command = validation.command ?? t('agent.validationNoCommand');
|
|
255
|
-
const detail = validation.status === 'skipped' && validation.skipReason
|
|
256
|
-
? `${validation.status}: ${validation.skipReason}`
|
|
257
|
-
: validation.status;
|
|
258
|
-
const symbol = validation.status === 'passed' ? '●' : validation.status === 'failed' ? '×' : '!';
|
|
259
|
-
layout.contentWrite(` ${color}${symbol}${ui.reset} ${t('agent.validationResult', { command, status: detail })}\n\n`);
|
|
260
|
-
},
|
|
261
244
|
onDone: (elapsedMs, usage) => {
|
|
262
245
|
flushToolBatch();
|
|
263
246
|
writeChangeOverview();
|
|
@@ -281,9 +264,7 @@ onContextUpdate) {
|
|
|
281
264
|
userInput,
|
|
282
265
|
signal,
|
|
283
266
|
onContextUpdate,
|
|
284
|
-
dynamicSystemSuffix: buildActiveNotesPlanReminder,
|
|
285
267
|
hooks: combinedHooks,
|
|
286
|
-
autoValidate: config.autoValidate,
|
|
287
268
|
onTraceEvent: appendCurrentSessionTraceEvent,
|
|
288
269
|
});
|
|
289
270
|
}
|
package/dist/agent/spawn.js
CHANGED
|
@@ -31,8 +31,8 @@ const SUBAGENT_ROLE = `## Sub-agent execution
|
|
|
31
31
|
You are executing one delegated sub-task with the same engineering standards and capabilities as mocode.
|
|
32
32
|
- Treat Task context as authoritative facts already established by the main agent; do not rediscover them without evidence they are stale.
|
|
33
33
|
- Focus on the delegated scope, but continue until it is genuinely complete. Do not stop to save tokens.
|
|
34
|
-
- Do not recursively call sub-agent. A write task runs in an isolated overlay; the coordinator merges
|
|
35
|
-
- Return concise findings, changes,
|
|
34
|
+
- Do not recursively call sub-agent. A write task runs in an isolated overlay; the coordinator merges it safely.
|
|
35
|
+
- Return concise findings, changes, checks you chose to run, and blockers to the coordinator.`;
|
|
36
36
|
/**
|
|
37
37
|
* 派生一个子 agent 执行独立子任务。
|
|
38
38
|
*
|
|
@@ -51,7 +51,7 @@ export async function spawnAgent(opts) {
|
|
|
51
51
|
summary: null,
|
|
52
52
|
completed: false,
|
|
53
53
|
transcript: 'Sub-agent execution is disabled. Enable it with /subagent on.',
|
|
54
|
-
status: 'failed', findings: [], readSet: [], changeSet: null,
|
|
54
|
+
status: 'failed', findings: [], readSet: [], changeSet: null,
|
|
55
55
|
usage: { promptTokens: 0, completionTokens: 0, totalTokens: 0, cachedTokens: 0, reasoningTokens: 0 },
|
|
56
56
|
};
|
|
57
57
|
}
|
|
@@ -135,7 +135,6 @@ export async function spawnAgent(opts) {
|
|
|
135
135
|
maxSteps,
|
|
136
136
|
toolsOverride,
|
|
137
137
|
contextState: localContextState,
|
|
138
|
-
autoValidate: false,
|
|
139
138
|
onToolOutcome: (tool, args) => {
|
|
140
139
|
if (tool === 'read_file' && typeof args.path === 'string')
|
|
141
140
|
readSet.add(args.path);
|
|
@@ -173,7 +172,6 @@ export async function spawnAgent(opts) {
|
|
|
173
172
|
findings: result.finalText ? [result.finalText] : [],
|
|
174
173
|
readSet: [...readSet].sort(),
|
|
175
174
|
changeSet,
|
|
176
|
-
verification: null, // 主 Agent 在所有 coordinator merge 完成后统一验证
|
|
177
175
|
usage: {
|
|
178
176
|
promptTokens: result.usage?.promptTokens ?? 0,
|
|
179
177
|
completionTokens: result.usage?.completionTokens ?? 0,
|
|
@@ -1,18 +1,5 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
// 把"完成必须验证"作为 system prompt 的一等公民,而不是事后选项。注入一段
|
|
4
|
-
// 4 阶段纪律(Plan & Discover → Build → Verify → Fix),并提供 per-model
|
|
5
|
-
// 措辞(anthropic / openai / qwen)让 base model 拿到最适合自己的表述。
|
|
6
|
-
//
|
|
7
|
-
// 关键约束:
|
|
8
|
-
// - 纯函数,无副作用,无 config 依赖 → 不踩 TDZ,易测,易回滚。
|
|
9
|
-
// - 段标题在 buildMocodeCorePrompt 之外,不会被 `## Project context` 索引
|
|
10
|
-
// 切片误伤;且 buildBasePrompt 注入位置在 ## Workflow 之前,确保 LLM
|
|
11
|
-
// 先看到纪律再看工具/平台细节。
|
|
12
|
-
// - per-model 措辞是"轻量"差异:3 个家族共享 4 阶段结构,只在首句
|
|
13
|
-
// 上贴近该家族的指令遵从习惯;真正的 prompt 反演化交给 AHE。
|
|
14
|
-
// - 语种统一英文:4 份都用同一份核心纪律文本,避免多语种漂移;用户语言
|
|
15
|
-
// 偏好由现有 i18n 段(assistant.languageInstruction)负责。
|
|
1
|
+
// Lightweight, advisory working guidance. The agent decides how much discovery and
|
|
2
|
+
// validation each task needs; the framework does not enforce a completion gate.
|
|
16
3
|
/**
|
|
17
4
|
* 从 config.model 字符串里嗅探 model family。匹配规则尽量宽松,够用即可。
|
|
18
5
|
* 未来 AHE 闭环后可以换成 config.modelFamily 字段。
|
|
@@ -33,73 +20,32 @@ export function inferModelFamily(model) {
|
|
|
33
20
|
* 4 阶段核心纪律(英文)。4 个 model family 共用此文本,只在首句与标题
|
|
34
21
|
* 标签上做轻量变体。保持短小,详细的完成检查由动态 checklist 按需注入。
|
|
35
22
|
*/
|
|
36
|
-
const CORE_SECTION = `## Working discipline — coding tasks
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
-
|
|
42
|
-
-
|
|
43
|
-
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
### Phase 3 — Verify
|
|
51
|
-
- Run the smallest executable check that proves the requested behavior, then read its complete result.
|
|
52
|
-
- Compare evidence with the user's request, not merely with the diff.
|
|
53
|
-
|
|
54
|
-
### Phase 4 — Fix
|
|
55
|
-
- Diagnose the root cause, make a focused correction, and rerun the relevant check.
|
|
56
|
-
- After two identical failures, change the approach instead of repeating the same call.
|
|
57
|
-
|
|
58
|
-
**Hard rule (non-negotiable):** "I read the code and it looks right" is not a completion signal. Report the verification performed, or state clearly why it could not be run.
|
|
59
|
-
|
|
60
|
-
**Hard rule (non-negotiable):** Never invent file paths, APIs, config keys, flags, or behavior. Every claim about the codebase must trace to tool output in this conversation; explicitly label anything you have not verified as an assumption.`;
|
|
61
|
-
/**
|
|
62
|
-
* 把核心段适配到指定 model family:只替换首行(语序 / 强动词),段标题
|
|
63
|
-
* 保持原样。Phase 内容保持原样,4 份共享同一份结构化文本。
|
|
64
|
-
* 注意:不再往标题注入 "[model: X]" 标签——它对模型是无意义噪声,
|
|
65
|
-
* 还可能引发自我指涉,反而干扰遵从。
|
|
66
|
-
*/
|
|
67
|
-
function adapt(_model, opener) {
|
|
68
|
-
return CORE_SECTION.replace('Treat "verification" as a first-class part of the task, not an afterthought.', opener);
|
|
69
|
-
}
|
|
23
|
+
const CORE_SECTION = `## Working discipline — coding tasks
|
|
24
|
+
|
|
25
|
+
Use your judgment to choose the shortest reliable path from the request to a useful result.
|
|
26
|
+
|
|
27
|
+
- Inspect only the code and context needed for the next decision.
|
|
28
|
+
- Make the smallest coherent change and avoid unrelated refactors.
|
|
29
|
+
- Decide whether validation is useful based on risk, scope, available commands, and the user's request. Validation is optional, not a completion gate.
|
|
30
|
+
- When validation is useful, choose the smallest relevant check yourself; do not run broad test/build suites by default.
|
|
31
|
+
- Re-read or rerun only when evidence is stale or the next edit depends on exact current content.
|
|
32
|
+
- On failure, diagnose before retrying; after repeated identical failures, change approach.
|
|
33
|
+
- Report honestly what you changed, what you checked, and anything left uncertain.
|
|
34
|
+
|
|
35
|
+
Never invent file paths, APIs, config keys, flags, or behavior. Distinguish repository evidence from assumptions.`;
|
|
70
36
|
/** ASK-01: only user-owned, high-impact choices should interrupt autonomous execution. */
|
|
71
|
-
const ASK_WHITELIST_SECTION = `## When to ask instead of guess
|
|
72
|
-
|
|
73
|
-
Call \`ask_human\` before coding only when repository evidence cannot resolve a user-owned, high-impact choice:
|
|
74
|
-
1. irreversible deletion, migration, security, permission, or external side effect;
|
|
75
|
-
2. public API compatibility (keep, deprecate, rename, or remove);
|
|
76
|
-
3. multiple reasonable options that materially change product behavior;
|
|
77
|
-
4. the request itself admits two or more materially different readings that lead to different deliverables (do not silently pick one and guess).
|
|
78
|
-
|
|
79
|
-
For naming
|
|
80
|
-
|
|
37
|
+
const ASK_WHITELIST_SECTION = `## When to ask instead of guess
|
|
38
|
+
|
|
39
|
+
Call \`ask_human\` before coding only when repository evidence cannot resolve a user-owned, high-impact choice:
|
|
40
|
+
1. irreversible deletion, migration, security, permission, or external side effect;
|
|
41
|
+
2. public API compatibility (keep, deprecate, rename, or remove);
|
|
42
|
+
3. multiple reasonable options that materially change product behavior;
|
|
43
|
+
4. the request itself admits two or more materially different readings that lead to different deliverables (do not silently pick one and guess).
|
|
44
|
+
|
|
45
|
+
For naming and implementation details, follow repository precedent and choose the safest reversible default. Disclose any consequential assumption.
|
|
46
|
+
|
|
81
47
|
Budget: at most 2 \`ask_human\` calls per turn. Beyond that, use the safest reversible default and disclose it in the final reply.`;
|
|
82
|
-
/**
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
* ASK-01 段是固定英文,不参与 per-model 适配(避免 4 份变体维护成本)。
|
|
86
|
-
*/
|
|
87
|
-
export function buildWorkDisciplineSection(modelFamily) {
|
|
88
|
-
let section;
|
|
89
|
-
switch (modelFamily) {
|
|
90
|
-
case 'anthropic':
|
|
91
|
-
section = adapt('anthropic', 'Verification is a hard prerequisite for completion, not a courtesy.');
|
|
92
|
-
break;
|
|
93
|
-
case 'openai':
|
|
94
|
-
section = adapt('openai', 'Every coding task MUST complete these four phases in order. Skipping or merging phases is treated as a failure. For trivial or read-only requests, phases may collapse.');
|
|
95
|
-
break;
|
|
96
|
-
case 'qwen':
|
|
97
|
-
section = adapt('qwen', 'Verification is a hard prerequisite for completion; "I wrote the code" is not evidence the code works.');
|
|
98
|
-
break;
|
|
99
|
-
case 'other':
|
|
100
|
-
case undefined:
|
|
101
|
-
default:
|
|
102
|
-
section = CORE_SECTION;
|
|
103
|
-
}
|
|
104
|
-
return `${section}\n\n${ASK_WHITELIST_SECTION}`;
|
|
48
|
+
/** Advisory guidance shared by main and sub-agents. */
|
|
49
|
+
export function buildWorkDisciplineSection(_modelFamily) {
|
|
50
|
+
return `${CORE_SECTION}\n\n${ASK_WHITELIST_SECTION}`;
|
|
105
51
|
}
|
package/dist/config/index.js
CHANGED
|
@@ -41,6 +41,7 @@ export const languageFromShell = process.env.MOCODE_LANGUAGE !== undefined;
|
|
|
41
41
|
// 仿 themeFromShell 模式:shell export 的环境变量在 loadEnvFiles 中不被回填(优先级最高),
|
|
42
42
|
// 故 /model 写入 ~/.mocode/config 的同名键下次启动会被 shell 值覆盖——据此给 dim 警告。
|
|
43
43
|
const LLM_ENV_KEYS = ['LLM_BASE_URL', 'LLM_API_KEY', 'LLM_MODEL', 'CONTEXT_WINDOW_TOKENS'];
|
|
44
|
+
export const DEFAULT_CONTEXT_WINDOW_TOKENS = 256000;
|
|
44
45
|
const llmKeysFromShell = LLM_ENV_KEYS.filter((k) => process.env[k] !== undefined);
|
|
45
46
|
loadEnvFiles();
|
|
46
47
|
setLanguage(detectLanguage(process.env.MOCODE_LANGUAGE));
|
|
@@ -195,11 +196,12 @@ ${buildWorkDisciplineSection(inferModelFamily(config.model))}
|
|
|
195
196
|
|
|
196
197
|
## Workflow
|
|
197
198
|
- Use existing conversation and tool evidence before gathering more. Inspect only what supports the next decision; do not guess.
|
|
198
|
-
- Keep changes focused.
|
|
199
|
+
- Keep changes focused. Decide for yourself whether a check is worth running; prefer the smallest relevant check and avoid broad test/build suites unless the task or risk justifies them.
|
|
199
200
|
- Use web search only when freshness materially affects the answer.
|
|
200
201
|
${buildCodegraphSection()}
|
|
201
202
|
|
|
202
203
|
## Tool use
|
|
204
|
+
- During tool-calling turns, stay silent unless something important enough must reach the user — otherwise just call the tool and let it run.
|
|
203
205
|
- Go directly to a known path or symbol; use discovery tools only when the location is unknown.
|
|
204
206
|
- Edit against a FRESH read: before any edit_file/write_file, call read_file on the exact path and copy both its latest hash and the exact target text. Never reconstruct old_string from a grep/summary/diff — those lose whitespace and indentation and cause edit failures.
|
|
205
207
|
- A read_file hash from before a compaction, session resume, edit conflict, or external change is STALE and will be rejected — re-read rather than reuse an old hash.
|
|
@@ -297,16 +299,14 @@ export const config = {
|
|
|
297
299
|
get systemPrompt() {
|
|
298
300
|
return buildBasePrompt();
|
|
299
301
|
},
|
|
300
|
-
contextWindowTokens: Number(process.env.CONTEXT_WINDOW_TOKENS) ||
|
|
301
|
-
compactThreshold: Number(process.env.COMPACT_THRESHOLD) || 0.85,
|
|
302
|
+
contextWindowTokens: Number(process.env.CONTEXT_WINDOW_TOKENS) || DEFAULT_CONTEXT_WINDOW_TOKENS,
|
|
302
303
|
includeUsage: process.env.LLM_STREAM_USAGE !== 'false',
|
|
303
304
|
autoCompact: process.env.AUTO_COMPACT !== 'false',
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
contextRelprune: process.env.MOCODE_CONTEXT_RELPRUNE !== 'false',
|
|
305
|
+
contextOptimize: process.env.MOCODE_CONTEXT_OPTIMIZE === 'true',
|
|
306
|
+
contextRelprune: process.env.MOCODE_CONTEXT_RELPRUNE === 'true',
|
|
307
307
|
contextLifecycle: process.env.MOCODE_LIFECYCLE !== 'false',
|
|
308
308
|
contextBudget: process.env.MOCODE_BUDGET_SCHEDULER !== 'false',
|
|
309
|
-
autoReflect: process.env.AUTO_REFLECT
|
|
309
|
+
autoReflect: process.env.AUTO_REFLECT === 'true',
|
|
310
310
|
memoryEnabled: process.env.MEMORY_ENABLED === 'true',
|
|
311
311
|
reflectEveryN: Number(process.env.REFLECT_EVERY_N) || 5,
|
|
312
312
|
maxSteps: Number(process.env.MAX_STEPS) || 1000,
|
|
@@ -1,12 +1,8 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
import { TOOL_OLD_AGE } from './budget.js';
|
|
1
|
+
// Pressure-only tool-result encoding coordinator.
|
|
2
|
+
// Normal pushes are raw apart from the per-result hard safety cap.
|
|
4
3
|
import { optimizeToolResult } from './pipeline.js';
|
|
5
4
|
import { canonicalizePath, extractPath, isToolResultSuccess, toText, } from './utils.js';
|
|
6
|
-
/**
|
|
7
|
-
* Tracks successful first reads and tool-result age without coupling encoders to
|
|
8
|
-
* lifecycle's mutable history indexes. All methods are fail-safe and idempotent.
|
|
9
|
-
*/
|
|
5
|
+
/** Rebuilds records from current history for each pressure pass. */
|
|
10
6
|
export class AgeAwareEncodingState {
|
|
11
7
|
pushOrdinal = 0;
|
|
12
8
|
records = new Map();
|
|
@@ -14,36 +10,16 @@ export class AgeAwareEncodingState {
|
|
|
14
10
|
constructor(history = []) {
|
|
15
11
|
this.rehydrate(history);
|
|
16
12
|
}
|
|
17
|
-
/**
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
this.records.set(tc.id, {
|
|
24
|
-
toolCallId: tc.id,
|
|
25
|
-
toolName: tc.name,
|
|
26
|
-
argsRaw: tc.arguments,
|
|
27
|
-
pushOrdinal: this.pushOrdinal,
|
|
28
|
-
succeeded,
|
|
29
|
-
isFirstRead,
|
|
30
|
-
agedEncoded: false,
|
|
31
|
-
});
|
|
32
|
-
this.pushOrdinal++;
|
|
33
|
-
// Failed reads must not consume the "first successful read" privilege.
|
|
34
|
-
if (succeeded && path)
|
|
35
|
-
this.seenReadPaths.add(path);
|
|
36
|
-
return {
|
|
37
|
-
age: 0,
|
|
38
|
-
isCold: false,
|
|
39
|
-
isFirstRead,
|
|
40
|
-
phase: 'push',
|
|
41
|
-
};
|
|
42
|
-
}
|
|
43
|
-
/** Re-encode eligible tool messages in the Cold prefix in place. */
|
|
44
|
-
sweep(history, hotBoundary) {
|
|
13
|
+
/**
|
|
14
|
+
* Pressure-only, progressive encoding of Cold logs and retrievable searches.
|
|
15
|
+
* It never touches code reads, skills, human decisions, or sub-agent output;
|
|
16
|
+
* repeated pressure passes may further reduce content only when strictly shorter.
|
|
17
|
+
*/
|
|
18
|
+
sweepPressure(history, hotBoundary) {
|
|
45
19
|
try {
|
|
20
|
+
const pressureEncodable = new Set(['run_command', 'grep', 'glob', 'web_search', 'web_fetch']);
|
|
46
21
|
const end = Math.min(Math.max(hotBoundary, 1), history.length);
|
|
22
|
+
let encodedCount = 0;
|
|
47
23
|
for (let idx = 1; idx < end; idx++) {
|
|
48
24
|
const message = history[idx];
|
|
49
25
|
if (message.role !== 'tool')
|
|
@@ -51,32 +27,27 @@ export class AgeAwareEncodingState {
|
|
|
51
27
|
const toolMessage = message;
|
|
52
28
|
const id = toolMessage.tool_call_id;
|
|
53
29
|
const record = id ? this.records.get(id) : undefined;
|
|
54
|
-
if (!record || !record.succeeded || record.
|
|
30
|
+
if (!record || !record.succeeded || !pressureEncodable.has(record.toolName))
|
|
55
31
|
continue;
|
|
56
32
|
const content = toText(toolMessage.content);
|
|
57
|
-
if (!content || content.startsWith('⌦['))
|
|
58
|
-
record.agedEncoded = true;
|
|
33
|
+
if (!content || content.startsWith('⌦['))
|
|
59
34
|
continue;
|
|
60
|
-
}
|
|
61
|
-
// Exclude the result's own push: immediately after insertion its age is 0.
|
|
62
35
|
const age = Math.max(0, this.pushOrdinal - record.pushOrdinal - 1);
|
|
63
|
-
if (age < TOOL_OLD_AGE)
|
|
64
|
-
continue;
|
|
65
36
|
const encoded = optimizeToolResult(record.toolName, content, record.argsRaw, {
|
|
66
37
|
age,
|
|
67
38
|
isCold: true,
|
|
68
39
|
isFirstRead: record.isFirstRead,
|
|
69
40
|
phase: 'sweep',
|
|
70
41
|
});
|
|
71
|
-
|
|
72
|
-
// representation that is equal-sized or larger.
|
|
73
|
-
if (encoded.length < content.length)
|
|
42
|
+
if (encoded.length < content.length) {
|
|
74
43
|
toolMessage.content = encoded;
|
|
75
|
-
|
|
44
|
+
encodedCount++;
|
|
45
|
+
}
|
|
76
46
|
}
|
|
47
|
+
return encodedCount;
|
|
77
48
|
}
|
|
78
49
|
catch {
|
|
79
|
-
|
|
50
|
+
return 0;
|
|
80
51
|
}
|
|
81
52
|
}
|
|
82
53
|
/** Rebuild stable state after resume or structural history compaction. */
|
|
@@ -119,7 +90,6 @@ export class AgeAwareEncodingState {
|
|
|
119
90
|
pushOrdinal: this.pushOrdinal,
|
|
120
91
|
succeeded,
|
|
121
92
|
isFirstRead,
|
|
122
|
-
agedEncoded: false,
|
|
123
93
|
});
|
|
124
94
|
this.pushOrdinal++;
|
|
125
95
|
if (succeeded && path)
|
|
@@ -132,10 +132,17 @@ export function recordArtifact(state, history, idx, output, succeeded) {
|
|
|
132
132
|
updateStats(state, stateFor(state));
|
|
133
133
|
}
|
|
134
134
|
function affected(artifact, changed) {
|
|
135
|
-
|
|
135
|
+
// '*' 依赖(无法解析出具体文件路径的诊断/搜索结果)不与任何具体写操作关联:
|
|
136
|
+
// 任何文件写入都会作废全部 '*' artifact,等于每次 mutation 都销毁
|
|
137
|
+
// git/测试/构建等历史证据,模型被迫反复 re-run,轮次爆炸。只失效路径明确命中的。
|
|
138
|
+
return artifact.dependencies.some((dependency) => dependency.path !== '*' && changed.has(dependency.path));
|
|
136
139
|
}
|
|
137
|
-
/**
|
|
138
|
-
|
|
140
|
+
/**
|
|
141
|
+
* Mark precise, file-backed facts stale without changing the evidence in history.
|
|
142
|
+
* A stale result can still explain a later edit or failure; pressure compression is
|
|
143
|
+
* the only path allowed to replace its content with a compact marker.
|
|
144
|
+
*/
|
|
145
|
+
export function invalidateArtifacts(state, _history, changedFiles) {
|
|
139
146
|
const changed = new Set(changedFiles.map(canonicalizePath).filter((item) => !!item));
|
|
140
147
|
if (changed.size === 0)
|
|
141
148
|
return 0;
|
|
@@ -145,15 +152,6 @@ export function invalidateArtifacts(state, history, changedFiles) {
|
|
|
145
152
|
if (artifact.freshness !== 'fresh' || !affected(artifact, changed))
|
|
146
153
|
continue;
|
|
147
154
|
artifact.freshness = 'stale';
|
|
148
|
-
const message = history[artifact.messageIndex];
|
|
149
|
-
if (message?.role === 'tool') {
|
|
150
|
-
const original = toText(message.content);
|
|
151
|
-
const paths = artifact.dependencies.map((item) => item.path).join(', ');
|
|
152
|
-
const stub = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id} dependencies=${paths} ` +
|
|
153
|
-
`invalidated-by=${[...changed].join(', ')}; re-run ${artifact.source.tool} before using this fact.`;
|
|
154
|
-
message.content = stub;
|
|
155
|
-
artifact.tokenCount = estimateTokens(stub);
|
|
156
|
-
}
|
|
157
155
|
count++;
|
|
158
156
|
}
|
|
159
157
|
updateStats(state, artifactState);
|
|
@@ -193,7 +191,7 @@ export function rehydrateArtifacts(state, history) {
|
|
|
193
191
|
}
|
|
194
192
|
updateStats(state, artifactState);
|
|
195
193
|
}
|
|
196
|
-
/** Compare captured dependency versions before each model step
|
|
194
|
+
/** Compare captured dependency versions before each model step and mark stale metadata only. */
|
|
197
195
|
export function refreshArtifactFreshness(state, history) {
|
|
198
196
|
const changed = new Set();
|
|
199
197
|
for (const artifact of stateFor(state).artifacts.values()) {
|
|
@@ -208,19 +206,23 @@ export function refreshArtifactFreshness(state, history) {
|
|
|
208
206
|
}
|
|
209
207
|
return changed.size > 0 ? invalidateArtifacts(state, history, [...changed]) : 0;
|
|
210
208
|
}
|
|
211
|
-
/**
|
|
212
|
-
|
|
209
|
+
/**
|
|
210
|
+
* Pressure-only stage: replace stale evidence in the Cold prefix with a compact
|
|
211
|
+
* marker. Recent/current work stays intact, and normal mutation handling never
|
|
212
|
+
* calls this function.
|
|
213
|
+
*/
|
|
214
|
+
export function pruneStaleArtifacts(state, history, coldBoundary) {
|
|
213
215
|
const artifactState = stateFor(state);
|
|
214
216
|
let pruned = 0;
|
|
215
217
|
for (const artifact of artifactState.artifacts.values()) {
|
|
216
|
-
if (artifact.freshness !== 'stale')
|
|
218
|
+
if (artifact.freshness !== 'stale' || artifact.messageIndex >= coldBoundary)
|
|
217
219
|
continue;
|
|
218
220
|
const message = history[artifact.messageIndex];
|
|
219
221
|
if (message?.role !== 'tool')
|
|
220
222
|
continue;
|
|
221
223
|
const content = toText(message.content);
|
|
222
224
|
if (!content.startsWith(STALE_PREFIX)) {
|
|
223
|
-
message.content = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id};
|
|
225
|
+
message.content = `${STALE_PREFIX}${artifact.source.tool}] source=${artifact.id}; source changed, so this content may be outdated.`;
|
|
224
226
|
artifact.tokenCount = estimateTokens(String(message.content));
|
|
225
227
|
pruned++;
|
|
226
228
|
}
|
package/dist/context/budget.js
CHANGED
|
@@ -1,12 +1,8 @@
|
|
|
1
|
-
// 五区 Context Budget
|
|
1
|
+
// 五区 Context Budget accounting and the shared pressure threshold.
|
|
2
2
|
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
// push-time cap、pipeline、relevance、lifecycle 与 age-aware sweep 负责工具结果优化;
|
|
7
|
-
// scheduler 在这些处理完成后评估,不重复生成 Cold/Hot tool 压缩动作。
|
|
8
|
-
// 本文件保持叶子级,只依赖 ChatMessage / token estimate,具体执行由 session/scheduler.ts 完成。
|
|
9
|
-
// contextBudget 开关关闭时,agent/core.ts 退化为直接调用 maybeCompact。
|
|
3
|
+
// This module only estimates and reports. session/scheduler.ts owns the sole
|
|
4
|
+
// automatic rewrite sequence and starts all pressure cleanup plus compact_history
|
|
5
|
+
// when corrected or raw request occupancy reaches 80%.
|
|
10
6
|
import { chatTools, estimateMessagesTokens, estimateToolSchemaTokens, messageTokens, } from '../llm/index.js';
|
|
11
7
|
import { toText } from './utils.js';
|
|
12
8
|
/** 五区分账(占比对齐 CONTEXT_WINDOW)。顺序固定,便于遍历。 */
|
|
@@ -28,9 +24,8 @@ export const DEFAULT_BUDGET_POLICY = {
|
|
|
28
24
|
reserve: 0.05,
|
|
29
25
|
},
|
|
30
26
|
hotTurnWindow: 4,
|
|
31
|
-
toolOldAge: 2,
|
|
32
27
|
compactKeepRatio: 0.40,
|
|
33
|
-
|
|
28
|
+
pressureTriggerRatio: 0.80,
|
|
34
29
|
schedulerTargetRatio: 0.80,
|
|
35
30
|
estimateSafetyFactor: 1.05,
|
|
36
31
|
compactHeadroomTokens: 1500,
|
|
@@ -38,7 +33,6 @@ export const DEFAULT_BUDGET_POLICY = {
|
|
|
38
33
|
/** 兼容既有调用方的只读别名;配置只在 DEFAULT_BUDGET_POLICY 中维护。 */
|
|
39
34
|
export const BUDGET_RATIO = DEFAULT_BUDGET_POLICY.ratios;
|
|
40
35
|
export const HOT_TURN_WINDOW = DEFAULT_BUDGET_POLICY.hotTurnWindow;
|
|
41
|
-
export const TOOL_OLD_AGE = DEFAULT_BUDGET_POLICY.toolOldAge;
|
|
42
36
|
function msgTokens(m) {
|
|
43
37
|
// 与请求预估复用同一实现,避免角色结构开销、多模态和 tool_calls 在两个预算路径中漂移。
|
|
44
38
|
return messageTokens(m);
|
|
@@ -69,13 +63,17 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
|
|
|
69
63
|
}
|
|
70
64
|
// 校正后的 token 数:raw * correction,最小 1(raw > 0 时)。
|
|
71
65
|
const adj = (raw) => (raw > 0 ? Math.max(1, Math.round(raw * correction)) : 0);
|
|
66
|
+
// 裸总量(不乘 correction):硬闸用它判断,防止 correction 折扣否决真实溢出。
|
|
67
|
+
let rawTotal = 0;
|
|
72
68
|
const sysMsg = history[0];
|
|
73
69
|
const systemCosts = {
|
|
74
70
|
prompt: sysMsg ? msgTokens(sysMsg) : 0,
|
|
75
71
|
toolSchemas: estimateToolSchemaTokens(activeTools),
|
|
76
72
|
};
|
|
77
73
|
// 工具 schema 与 system prompt 同属请求固定开销;必须计入总量才能可靠触发压缩。
|
|
78
|
-
|
|
74
|
+
const systemRaw = systemCosts.prompt + systemCosts.toolSchemas;
|
|
75
|
+
layers.system.actual = adj(systemRaw);
|
|
76
|
+
rawTotal += systemRaw;
|
|
79
77
|
// Summary 检测:role:'system' 且不是 history[0] 的,视为摘要(compact.ts 摘要插 index 1)。
|
|
80
78
|
// 简单启发:若 history[1]?.role === 'system' 且 content 含「# 会话摘要」特征串,计入 summary。
|
|
81
79
|
// 命中时循环跳过 i=1;不命中时当作普通 message(罕见,落到下方 user/assistant 分支)。
|
|
@@ -83,7 +81,9 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
|
|
|
83
81
|
if (history.length > 1 && history[1].role === 'system') {
|
|
84
82
|
const c1 = toText(history[1].content);
|
|
85
83
|
if (c1.startsWith('# 会话摘要') || c1.includes('会话摘要')) {
|
|
86
|
-
|
|
84
|
+
const summaryRaw = msgTokens(history[1]);
|
|
85
|
+
layers.summary.actual = adj(summaryRaw);
|
|
86
|
+
rawTotal += summaryRaw;
|
|
87
87
|
summaryHit = true;
|
|
88
88
|
}
|
|
89
89
|
}
|
|
@@ -94,16 +94,18 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
|
|
|
94
94
|
const m = history[i];
|
|
95
95
|
if (i === 1 && summaryHit)
|
|
96
96
|
continue; // summary 已单独算过
|
|
97
|
+
const raw = msgTokens(m);
|
|
97
98
|
if (m.role === 'tool') {
|
|
98
|
-
const t = adj(msgTokens(m));
|
|
99
99
|
if (i >= hotStart)
|
|
100
|
-
layers.toolRecent.actual +=
|
|
100
|
+
layers.toolRecent.actual += adj(raw);
|
|
101
101
|
else
|
|
102
|
-
layers.toolOld.actual +=
|
|
102
|
+
layers.toolOld.actual += adj(raw);
|
|
103
|
+
rawTotal += raw;
|
|
103
104
|
}
|
|
104
105
|
else if (m.role !== 'system') {
|
|
105
106
|
// user / assistant 全部计入 history(对话轨迹)
|
|
106
|
-
layers.history.actual += adj(
|
|
107
|
+
layers.history.actual += adj(raw);
|
|
108
|
+
rawTotal += raw;
|
|
107
109
|
}
|
|
108
110
|
// 其它 system(几乎不存在)跳过
|
|
109
111
|
}
|
|
@@ -120,10 +122,11 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
|
|
|
120
122
|
// 按 overRatio 降序
|
|
121
123
|
triggers.sort((a, b) => layers[b].overRatio - layers[a].overRatio);
|
|
122
124
|
const total = BUDGET_LAYERS.reduce((s, k) => s + (k === 'reserve' ? 0 : layers[k].actual), 0);
|
|
123
|
-
const totalOver = total >= DEFAULT_BUDGET_POLICY.
|
|
125
|
+
const totalOver = total >= DEFAULT_BUDGET_POLICY.pressureTriggerRatio * window;
|
|
124
126
|
return {
|
|
125
127
|
step,
|
|
126
128
|
total,
|
|
129
|
+
rawTotal,
|
|
127
130
|
window,
|
|
128
131
|
layers,
|
|
129
132
|
systemCosts,
|
|
@@ -134,15 +137,11 @@ export function evaluateBudget(history, window, step = 0, correction = 1, active
|
|
|
134
137
|
};
|
|
135
138
|
}
|
|
136
139
|
/** 根据 BudgetReport 生成可执行动作。
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
* History 或总量超预算时才考虑昂贵的 LLM 摘要。 */
|
|
140
|
+
* History compaction is the final fallback after pressure-only tool stages.
|
|
141
|
+
* Per-layer overages are diagnostics, never independent rewrite triggers. */
|
|
140
142
|
export function scheduleActions(report) {
|
|
141
143
|
const actions = [];
|
|
142
|
-
const { layers
|
|
143
|
-
const policy = DEFAULT_BUDGET_POLICY;
|
|
144
|
-
const headroom = policy.schedulerTargetRatio * report.window
|
|
145
|
-
- total * policy.estimateSafetyFactor;
|
|
144
|
+
const { layers } = report;
|
|
146
145
|
if (layers.system.overBudget) {
|
|
147
146
|
const { prompt, toolSchemas } = report.systemCosts;
|
|
148
147
|
const { actual, budget } = layers.system;
|
|
@@ -155,8 +154,8 @@ export function scheduleActions(report) {
|
|
|
155
154
|
+ `提示 ${prompt} + 工具 ${toolSchemas},×${report.correction.toFixed(2)}。`,
|
|
156
155
|
});
|
|
157
156
|
}
|
|
158
|
-
|
|
159
|
-
|
|
157
|
+
const pressureLine = DEFAULT_BUDGET_POLICY.pressureTriggerRatio * report.window;
|
|
158
|
+
if (Math.max(report.rawTotal, report.total) >= pressureLine) {
|
|
160
159
|
actions.push({ kind: 'compact_history' });
|
|
161
160
|
}
|
|
162
161
|
return actions;
|
|
@@ -164,7 +163,7 @@ export function scheduleActions(report) {
|
|
|
164
163
|
/** 拍平成人类可读(供 /context 命令与 check-budget 脚本用)。 */
|
|
165
164
|
export function formatReport(report) {
|
|
166
165
|
const lines = [];
|
|
167
|
-
lines.push(`step ${report.step} total ${report.total}/${report.window} (${((report.total / report.window) * 100).toFixed(1)}%)`);
|
|
166
|
+
lines.push(`step ${report.step} total ${report.total}/${report.window} (${((report.total / report.window) * 100).toFixed(1)}%) raw ${report.rawTotal}`);
|
|
168
167
|
for (const k of BUDGET_LAYERS) {
|
|
169
168
|
const lb = report.layers[k];
|
|
170
169
|
const pct = lb.budget > 0 ? ((lb.actual / lb.budget) * 100).toFixed(0) : '-';
|
|
@@ -1,14 +1,7 @@
|
|
|
1
|
-
// 内置 encoder
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
// Phase 2.5:code / graph / doc / summary(覆盖剩余有 encoder 的 kind)。
|
|
6
|
-
// - code(read_file):仅折叠 ≥3 连续空行,行号保真(edit_file 依赖)。
|
|
7
|
-
// - graph(保留 kind,暂无 builtin 工具直接命中)/ doc(web_fetch, use_skill)/ summary(task):
|
|
8
|
-
// 去 ANSI + 折叠空行,保守不重构结构。
|
|
9
|
-
// - status(edit/write/ask_human/mem 增删改)无 encoder:本就是一行,无需编码。
|
|
10
|
-
//
|
|
11
|
-
// 加 encoder:新建 encoders/xxx.ts 导出 ContextEncoder,在此数组加一行。无需动 agent / llm / core。
|
|
1
|
+
// 内置 encoder 清单。仅由 opt-in pressure stage 首次调用时懒注册。
|
|
2
|
+
// Normal tool pushes do not pass through these encoders.
|
|
3
|
+
// Available transforms:tree / search / log / table / memory / code / doc / summary.
|
|
4
|
+
// status 类工具本就是短结果,无专用 encoder。
|
|
12
5
|
import { passthroughEncoder } from './passthrough.js';
|
|
13
6
|
import { treeEncoder } from './tree.js';
|
|
14
7
|
import { searchEncoder } from './search.js';
|
package/dist/context/index.js
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
|
-
// context/ barrel:
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
// 单一入口 runScheduler(agent/core.ts 步前调)接管"何时调用哪一闸"的调度。
|
|
5
|
-
// 不调 LLM、不碰 Tool Calling schema / executeTool / tool_call_id 配对 / TUI 渲染
|
|
6
|
-
// (叶子级:仅 stdlib + tools/constants + session/compact 的 capToolResultForHistory 兜底 + config 开关)。
|
|
1
|
+
// context/ barrel: metadata tracking, optional pressure encoders, and budget reporting.
|
|
2
|
+
// Normal tool pushes stay raw apart from the hard per-result cap. The session
|
|
3
|
+
// scheduler is the only automatic rewrite coordinator at real pressure.
|
|
7
4
|
export { optimizeToolResult } from './pipeline.js';
|
|
8
5
|
export { classify, knownToolKinds } from './classifier.js';
|
|
9
6
|
export { recordArtifact, invalidateArtifacts, rehydrateArtifacts, refreshArtifactFreshness, pruneStaleArtifacts, collectArtifactRefs, formatArtifactTokenSources, } from './artifacts.js';
|
|
10
7
|
export { registerEncoder, registerAll, getEncoder, registeredKinds, } from './registry.js';
|
|
11
8
|
// ── Context Budget Scheduler ───────────────────────────────────────────────
|
|
12
|
-
export { evaluateBudget, scheduleActions, formatReport, quickEstimate, userTurnBoundary, BUDGET_LAYERS, DEFAULT_BUDGET_POLICY, BUDGET_RATIO, HOT_TURN_WINDOW,
|
|
9
|
+
export { evaluateBudget, scheduleActions, formatReport, quickEstimate, userTurnBoundary, BUDGET_LAYERS, DEFAULT_BUDGET_POLICY, BUDGET_RATIO, HOT_TURN_WINDOW, } from './budget.js';
|