mocode-ai 1.1.7 → 1.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -40
- package/README.zh-CN.md +51 -41
- package/dist/agent/core.js +137 -447
- package/dist/agent/index.js +5 -24
- package/dist/agent/spawn.js +5 -5
- package/dist/agent/work-discipline.js +16 -70
- package/dist/config/index.js +87 -33
- package/dist/context/age-aware.js +18 -48
- package/dist/context/artifacts.js +19 -17
- package/dist/context/budget.js +27 -28
- package/dist/context/classifier.js +0 -1
- package/dist/context/encoders/index.js +4 -11
- package/dist/context/index.js +4 -7
- package/dist/context/lifecycle.js +115 -483
- package/dist/context/pipeline.js +8 -15
- package/dist/context/relevance.js +77 -55
- package/dist/host/stdio.js +0 -6
- package/dist/i18n/index.js +20 -6
- package/dist/index.js +11 -1
- package/dist/llm/index.js +82 -9
- package/dist/mcp/index.js +0 -1
- package/dist/repl/index.js +69 -15
- package/dist/runtime/browser-manager.js +299 -0
- package/dist/runtime/dev-server-manager.js +354 -0
- package/dist/runtime/shutdown.js +26 -0
- package/dist/session/compact.js +86 -102
- package/dist/session/index.js +0 -1
- package/dist/session/notes.js +107 -0
- package/dist/session/scheduler.js +88 -92
- package/dist/session/trace-metrics.js +5 -92
- package/dist/session/trace.js +1 -10
- package/dist/tools/builtins/browser.js +199 -0
- package/dist/tools/builtins/dev-server.js +99 -0
- package/dist/tools/builtins/index.js +33 -19
- package/dist/tools/builtins/plan-update.js +144 -0
- package/dist/tools/builtins/screenshot.js +173 -0
- package/dist/tools/builtins/view-image.js +49 -0
- package/dist/tools/constants.js +38 -8
- package/dist/tools/registry.js +5 -29
- package/dist/ui/batch.js +3 -0
- package/dist/ui/content.js +19 -16
- package/dist/ui/layout.js +120 -46
- package/dist/ui/render.js +11 -0
- package/package.json +2 -2
- package/dist/agent/middleware/checklist.js +0 -59
- package/dist/session/drop.d.ts +0 -19
- package/dist/session/drop.js +0 -93
- package/dist/tools/builtins/drop-context.d.ts +0 -18
- package/dist/tools/builtins/drop-context.js +0 -68
- package/dist/verification/diagnostics.js +0 -108
- package/dist/verification/fingerprint.js +0 -54
- package/dist/verification/index.js +0 -333
- package/dist/verification/postconditions.js +0 -98
- package/dist/verification/targeted-tests.js +0 -96
- package/dist/verification/types.js +0 -1
package/dist/agent/core.js
CHANGED
|
@@ -5,39 +5,32 @@
|
|
|
5
5
|
// 与 index.ts 的关系:index.ts 的 runAgent = runAgentCore + TUI hooks 薄封装(行为不变)。
|
|
6
6
|
// spawn.ts 的 spawnAgent = runAgentCore + 静默 hooks(子 agent)。
|
|
7
7
|
import { readFileSync } from 'node:fs';
|
|
8
|
+
import { getNotesMtime } from '../session/notes.js';
|
|
8
9
|
import { chat, estimatePromptTokens, planChatTools, chatTools, } from '../llm/index.js';
|
|
9
10
|
import { executeToolOutcome, getToolCapabilities, isFileMutationTool, tools, } from '../tools/registry.js';
|
|
10
11
|
import { checkPermission } from '../permissions/index.js';
|
|
11
12
|
import { validateToolArguments } from '../tools/validation.js';
|
|
12
13
|
import { getPlanDisabledTools, getRuntimeDisabledTools } from '../tools/constants.js';
|
|
13
|
-
import { createPreCompletionChecklistMiddleware, } from './middleware/checklist.js';
|
|
14
|
-
import { inferModelFamily } from './work-discipline.js';
|
|
15
|
-
import { reflectionHint, classifyError } from './retry-classifier.js';
|
|
16
14
|
import { getAgentMode, setAgentMode } from './mode.js';
|
|
17
|
-
import { maybeCompact, contextState,
|
|
15
|
+
import { maybeCompact, contextState, createTraceEvent, summarizeToolArguments, safeProviderId, } from '../session/index.js';
|
|
16
|
+
import { capToolResultForHistory } from '../session/compact.js';
|
|
18
17
|
import { createBudgetScheduler } from '../session/scheduler.js';
|
|
19
|
-
import {
|
|
20
|
-
import { createAgeAwareEncodingState, } from '../context/age-aware.js';
|
|
18
|
+
import { recordArtifact, invalidateArtifacts, rehydrateArtifacts, } from '../context/index.js';
|
|
21
19
|
import { createRelevancePruner } from '../context/relevance.js';
|
|
22
20
|
import { isToolResultSuccess } from '../context/utils.js';
|
|
23
|
-
import { config } from '../config/index.js';
|
|
21
|
+
import { config, extractActivePlanSection, reinjectActivePlanIntoSystem } from '../config/index.js';
|
|
24
22
|
import { t } from '../i18n/index.js';
|
|
25
23
|
import { jailResolve } from '../sandbox/index.js';
|
|
26
24
|
import { createLifecycleEngine } from '../context/lifecycle.js';
|
|
27
25
|
import { getTokenCalibration, updateTokenCalibration, } from '../context/token-calibration.js';
|
|
28
26
|
import { getCurrentTurnId, getCurrentTurnMutationState } from '../rollback/index.js';
|
|
29
|
-
import { createAutomaticValidator, } from '../verification/index.js';
|
|
30
27
|
import { getCurrentSessionId } from '../session/state.js';
|
|
31
|
-
/**
|
|
32
|
-
const
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
const created = createAgeAwareEncodingState(history);
|
|
38
|
-
ageAwareStateByHistory.set(history, created);
|
|
39
|
-
return created;
|
|
40
|
-
}
|
|
28
|
+
/** nag 提醒阈值:连续 N 个"执行了工具但没更新 notes.md"的步后提醒一次(对齐 Claude Code TodoWrite 的 3 轮)。 */
|
|
29
|
+
const PLAN_NAG_THRESHOLD = 3;
|
|
30
|
+
/** nag 提醒文本:注入到当前步第一条 tool_result 内容前(与最新工具输出同批被模型看到,而非单独一条易被冲淡)。 */
|
|
31
|
+
const PLAN_NAG_TEXT = '[mocode] Reminder: you have an active plan in notes.md but have not updated it recently. ' +
|
|
32
|
+
'If you finished a step, call plan_update to check it off (keep at most one in_progress); ' +
|
|
33
|
+
'if the whole plan is done, let plan_update settle it to ## Done:. If the plan changed scope, update it to match reality.';
|
|
41
34
|
/** 解析工具 arguments JSON;非法或空返 null(调用方据此降级到普通 preview)。 */
|
|
42
35
|
function parseArgs(raw) {
|
|
43
36
|
try {
|
|
@@ -47,49 +40,6 @@ function parseArgs(raw) {
|
|
|
47
40
|
return null;
|
|
48
41
|
}
|
|
49
42
|
}
|
|
50
|
-
/**
|
|
51
|
-
* Thrashing 检测:同一工具 + 完全相同 arguments 在本轮重复 ≥ THRASH_THRESHOLD 次,
|
|
52
|
-
* 返一段提示(注入到工具结果尾部),引导模型换思路而不是再试一次。
|
|
53
|
-
* 阈值 2 = "试过两次同样的调用还没好,该停了"。指纹 = `${name}\\x00${args}`
|
|
54
|
-
* (直接拼,不哈希——避免热路径开销;args 长度本身有限,内存压力可忽略)。
|
|
55
|
-
* null 表示未触发,不污染输出。
|
|
56
|
-
*/
|
|
57
|
-
const THRASH_THRESHOLD = 2;
|
|
58
|
-
function thrashHint(name, args, count) {
|
|
59
|
-
if (count < THRASH_THRESHOLD)
|
|
60
|
-
return null;
|
|
61
|
-
return (`\n\n[hint] This is call #${count} of \`${name}\` with identical arguments — ` +
|
|
62
|
-
'either failing or returning the same content. STOP retrying and switch strategy:\n' +
|
|
63
|
-
'- read_file / glob → path likely wrong; call `glob` to discover paths, or `ask_human`\n' +
|
|
64
|
-
'- run_command → Windows path-escaping issue; use `read_file` / `glob` with absolute paths instead\n' +
|
|
65
|
-
'- write_file / edit_file CHANGE_CONFLICT → do not resend; read_file the same path and use its latest hash (use null only when read_file says the path is missing)\n' +
|
|
66
|
-
'- edit_file old_string mismatch → re-read the exact region and copy it verbatim\n' +
|
|
67
|
-
'- otherwise → re-read the tool description; the argument shape may be wrong');
|
|
68
|
-
}
|
|
69
|
-
/**
|
|
70
|
-
* Detect only a strict streak of identical failures. Successful calls, or a
|
|
71
|
-
* different tool/argument pair, end the streak. This keeps intentional phased
|
|
72
|
-
* read_file/glob calls from being mislabeled after the underlying files change.
|
|
73
|
-
*/
|
|
74
|
-
export function createThrashTracker() {
|
|
75
|
-
let lastFailedFingerprint = null;
|
|
76
|
-
let consecutiveFailures = 0;
|
|
77
|
-
return (name, args, succeeded) => {
|
|
78
|
-
if (succeeded) {
|
|
79
|
-
lastFailedFingerprint = null;
|
|
80
|
-
consecutiveFailures = 0;
|
|
81
|
-
return null;
|
|
82
|
-
}
|
|
83
|
-
const fingerprint = `${name}\x00${args}`;
|
|
84
|
-
if (fingerprint === lastFailedFingerprint)
|
|
85
|
-
consecutiveFailures += 1;
|
|
86
|
-
else {
|
|
87
|
-
lastFailedFingerprint = fingerprint;
|
|
88
|
-
consecutiveFailures = 1;
|
|
89
|
-
}
|
|
90
|
-
return thrashHint(name, args, consecutiveFailures);
|
|
91
|
-
};
|
|
92
|
-
}
|
|
93
43
|
/** 只有显式声明 parallel 且无需权限确认的工具才进入普通并发组。 */
|
|
94
44
|
function isParallelTool(name) {
|
|
95
45
|
const tool = tools.find((candidate) => candidate.name === name);
|
|
@@ -158,86 +108,18 @@ function readDiffContext(tc, parsed) {
|
|
|
158
108
|
}
|
|
159
109
|
return { preWriteOld: null, editStartLine: 1 };
|
|
160
110
|
}
|
|
161
|
-
/** 回灌 tool 结果到 history
|
|
162
|
-
*
|
|
163
|
-
*
|
|
164
|
-
*
|
|
165
|
-
* - pruner 在每个 runAgentCore 实例化一次(本闭包持有),会话级状态。
|
|
166
|
-
* - 开关关闭时 pruner 不创建(零开销、零行为变化)。
|
|
167
|
-
* 出口再经 Lifecycle Engine 做引用追踪:LIVE→REFERENCED→OBSOLETE→STUB 四态。
|
|
168
|
-
* - lifecycle 也在每个 runAgentCore 实例化一次,登记 grep/glob/web_search/web_fetch 等 producer
|
|
169
|
-
* 与 read/edit/write 的 consumer 关系;孤立+老化自动 STUB(观察类工具永不到 STUB)。
|
|
170
|
-
* - 开关关闭时 lifecycle=null 完全跳过。 */
|
|
171
|
-
/** RETRY-01: 把 thrash hint + 反思指针一次性拼到 output 尾部。
|
|
172
|
-
* - thrash hint 永远可能存在(连续失败才出现);
|
|
173
|
-
* - 反思指针仅在 error/denied 状态追加,success 跳过避免噪声。
|
|
174
|
-
* 抽出 helper 是为了 sequential + parallel 两处共用,避免拼装逻辑漂移。
|
|
175
|
-
* 返回追加后的字符串 + 反思 category(用于 QUAL-01 trace 硬事件),
|
|
176
|
-
* success / aborted 时 category === null,调用方据此决定是否 emit 'retry_reflection' 事件。 */
|
|
177
|
-
function appendRetryAnnotations(output, status, code, thrashHint) {
|
|
178
|
-
let result = thrashHint ? `${output}${thrashHint}` : output;
|
|
179
|
-
let reflectionCategory = null;
|
|
180
|
-
if (status !== 'success' && status !== 'aborted') {
|
|
181
|
-
reflectionCategory = classifyError(code);
|
|
182
|
-
const reflection = reflectionHint(reflectionCategory);
|
|
183
|
-
result = `${result}\n\n[retry reflection: ${reflectionCategory}]\n${reflection}`;
|
|
184
|
-
}
|
|
185
|
-
return { output: result, reflectionCategory };
|
|
186
|
-
}
|
|
187
|
-
/** ASK-01: ask_human 工具调用本 turn 计数器。纯函数无副作用,
|
|
188
|
-
* 计数规则简单:1-based(本 turn 第 1 次 = 1, 第 2 次 = 2, 第 3+ 次 = exceeded)。
|
|
189
|
-
* 预算上限常量与工作纪律段 "Budget: at most 2 ask_human calls per turn" 保持一致;
|
|
190
|
-
* 改这里时,fixture 与纪律段都要同步。 */
|
|
191
|
-
export const ASK_HUMAN_PER_TURN_BUDGET = 2;
|
|
192
|
-
export function askHumanBudgetAnnotation(askHumanCountThisTurn, status) {
|
|
193
|
-
// 工具失败 / aborted 不追加(避免噪声;模型已被错误消息告知失败)。
|
|
194
|
-
if (status !== 'success')
|
|
195
|
-
return null;
|
|
196
|
-
if (askHumanCountThisTurn < ASK_HUMAN_PER_TURN_BUDGET)
|
|
197
|
-
return null;
|
|
198
|
-
if (askHumanCountThisTurn === ASK_HUMAN_PER_TURN_BUDGET) {
|
|
199
|
-
return `\n\n[ask budget] This was your ${askHumanCountThisTurn}nd ask_human call this turn (budget = ${ASK_HUMAN_PER_TURN_BUDGET}). For the rest of this turn, prefer the safer default and disclose the choice in your final reply — do not silently guess.`;
|
|
200
|
-
}
|
|
201
|
-
// 第 3+ 次(超过预算):强硬提示,鼓励模型停下问自己是否还有意义。
|
|
202
|
-
return `\n\n[ask budget EXCEEDED] This is ask_human call #${askHumanCountThisTurn} this turn (budget = ${ASK_HUMAN_PER_TURN_BUDGET}). Stop asking; choose a default, implement, and disclose the choice in your final reply. Continuing to ask is more harmful than a documented guess.`;
|
|
203
|
-
}
|
|
204
|
-
/** NARR-01: 工具轮旁白(assistant 消息同时带 content + tool_calls)的软预算。
|
|
205
|
-
* 超出即追加回压提示;未超只 emit trace(可度量,不打扰)。
|
|
206
|
-
* 单位是 code point 而非字节——中文一字一 point,英文一字母一 point,
|
|
207
|
-
* 对两种语言都是"一句话大约多长"的直觉量级。 */
|
|
208
|
-
export const NARRATION_CHAR_BUDGET = 120;
|
|
209
|
-
/** 分类工具轮旁白。返回 null 表示这一轮没有旁白(纯工具调用,理想情况)。
|
|
210
|
-
* hint 仅在超预算时非 null:prompt 里的"工具轮保持静默"是软约束,
|
|
211
|
-
* 这里给出机制层面的回压,同时让 trace 能统计旁白率。 */
|
|
212
|
-
export function classifyNarration(content, toolCallCount) {
|
|
213
|
-
const text = content?.trim() ?? '';
|
|
214
|
-
if (!text)
|
|
215
|
-
return null;
|
|
216
|
-
const chars = [...text].length;
|
|
217
|
-
if (chars <= NARRATION_CHAR_BUDGET) {
|
|
218
|
-
return { chars, overBudget: false, hint: null };
|
|
219
|
-
}
|
|
220
|
-
const plural = toolCallCount === 1 ? '' : 's';
|
|
221
|
-
return {
|
|
222
|
-
chars,
|
|
223
|
-
overBudget: true,
|
|
224
|
-
hint: `\n\n[narration] Your previous message emitted ${chars} characters of prose alongside ` +
|
|
225
|
-
`${toolCallCount} tool call${plural} (soft budget = ${NARRATION_CHAR_BUDGET}). ` +
|
|
226
|
-
'Interstitial commentary costs the user screen space and costs you context on every ' +
|
|
227
|
-
'later step. For the rest of this turn, emit prose mid-turn ONLY to (a) flag a genuine ' +
|
|
228
|
-
'decision fork needing user input, (b) disclose an error or risk, or (c) deliver the ' +
|
|
229
|
-
'final answer once tools are done. Otherwise call the next tool with no preamble.',
|
|
230
|
-
};
|
|
231
|
-
}
|
|
111
|
+
/** 回灌 tool 结果到 history。
|
|
112
|
+
* 正常路径只做单条 hard cap;原始 output 同时供 TUI 展示,因此用户与模型
|
|
113
|
+
* 看到同一事实。Artifact/Relevance/Lifecycle 仅登记 metadata/provenance,
|
|
114
|
+
* 不在这里改写旧正文;所有自动清理与压缩统一由 80% pressure scheduler 决定。 */
|
|
232
115
|
function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runtimeContextState = contextState, succeededOverride) {
|
|
233
116
|
const succeeded = succeededOverride ?? isToolResultSuccess(output);
|
|
234
|
-
const ageAware = config.contextOptimize ? ageAwareStateFor(history) : null;
|
|
235
|
-
const encodingContext = ageAware?.preparePush(tc, succeeded);
|
|
236
117
|
const msg = {
|
|
237
118
|
role: 'tool',
|
|
238
119
|
tool_call_id: tc.id,
|
|
239
|
-
//
|
|
240
|
-
|
|
120
|
+
// Preserve evidence verbatim in normal operation; the hard per-result cap
|
|
121
|
+
// remains solely as a request-size safety rail.
|
|
122
|
+
content: capToolResultForHistory(tc.name, output),
|
|
241
123
|
};
|
|
242
124
|
history.push(msg);
|
|
243
125
|
const messageIndex = history.length - 1;
|
|
@@ -253,8 +135,8 @@ function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runt
|
|
|
253
135
|
* agent 核心循环(纯逻辑):
|
|
254
136
|
* 流式调 LLM(经 hooks.onText 实时渲染)→ 有 tool_calls 就分组执行并回灌
|
|
255
137
|
* → 否则流式正文即最终回复。history 在调用间持久,由调用方持有。
|
|
256
|
-
*
|
|
257
|
-
*
|
|
138
|
+
* 步前由 session scheduler 检查真实 context pressure;达到 80% 时统一清理并压缩历史。
|
|
139
|
+
* 工具结果正常只经 capToolResultForHistory 的单条 hard safety cap。
|
|
258
140
|
*
|
|
259
141
|
* 中断语义:signal 经 executeTool(name, args, signal) 串进工具;run_command/web_fetch 等 abort 即时杀
|
|
260
142
|
* (树杀子进程 / 取消 fetch),循环顶 if(signal.aborted) 兜底还原。不会留下未配对的 tool_call_id。
|
|
@@ -267,25 +149,8 @@ function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runt
|
|
|
267
149
|
export async function runAgentCore(opts) {
|
|
268
150
|
const { history, userInput, signal, onContextUpdate, hooks } = opts;
|
|
269
151
|
const runtimeContextState = opts.contextState ?? contextState;
|
|
270
|
-
|
|
271
|
-
// 预算超过时,工具结果尾部追加 ask budget 提示,不直接拒绝调用(更轻量,
|
|
272
|
-
// 也避免和现有 permission 系统的"拒绝"语义重叠)。
|
|
152
|
+
/** 本轮 ask_human 成功调用次数,仅用于 trace 观测,不影响工具执行或模型上下文。 */
|
|
273
153
|
let askHumanCountThisTurn = 0;
|
|
274
|
-
// NARR-01: 上一条 assistant 消息(带 tool_calls)超出旁白预算时,待注入的回压提示。
|
|
275
|
-
// 只挂在本批次的第一条工具结果上并立即清空——重复注入会变成新的噪声源。
|
|
276
|
-
let pendingNarrationHint = null;
|
|
277
|
-
/** 取出并清空待注入的旁白回压提示(每批工具只消费一次)。 */
|
|
278
|
-
const takeNarrationHint = () => {
|
|
279
|
-
const hint = pendingNarrationHint;
|
|
280
|
-
pendingNarrationHint = null;
|
|
281
|
-
return hint;
|
|
282
|
-
};
|
|
283
|
-
// PROMPT-02: 解析 preCompletionChecklist 选项。undefined = 启用默认 middleware;
|
|
284
|
-
// false = opt-out(checklist 自身调试用);function = 调用方自定义。
|
|
285
|
-
const _checklistMiddleware = createPreCompletionChecklistMiddleware();
|
|
286
|
-
const preCompletionChecklistHandler = opts.preCompletionChecklist === false
|
|
287
|
-
? null
|
|
288
|
-
: (opts.preCompletionChecklist ?? _checklistMiddleware.handler);
|
|
289
154
|
const maxSteps = opts.maxSteps ?? config.maxSteps;
|
|
290
155
|
// 中断还原:repl 的 /plan / /auto / Shift+Tab 等用户面触发 setAgentMode 中途切了模式,
|
|
291
156
|
// abort 时连同模式一起还原回轮首。模型不再持有 switch_mode 工具,无法自切。
|
|
@@ -318,14 +183,17 @@ export async function runAgentCore(opts) {
|
|
|
318
183
|
let done = false; // 正常完毕 / 达上限 true;中断 false(不显摘要)
|
|
319
184
|
let traceStatus = 'error';
|
|
320
185
|
let toolCallCount = 0;
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
let
|
|
325
|
-
const validator = opts.validator ?? createAutomaticValidator();
|
|
186
|
+
// A+B(plan 可靠性):跨步计数"执行了工具但没改动 notes.md"的连续步数。
|
|
187
|
+
// 本步写了 notes.md(plan_update 或直接 write/edit)→ 清零并重同步 history[0];
|
|
188
|
+
// 否则累计,达阈值则在当前步 tool_result 前注入 nag 提醒。
|
|
189
|
+
let stepsSincePlanTouch = 0;
|
|
326
190
|
// 本轮 token 累计:每步 chat() 返回后把 result.usage 累加,供 onDone 摘要行 + AgentRunResult.usage
|
|
327
191
|
// 透传给 repl(显示在底栏模式 chip 右边)。未开启 include_usage 或全失败时为 undefined。
|
|
328
192
|
let turnUsage;
|
|
193
|
+
// 实时 chip ↻ 估算用:上一步 chat 实测 prompt(前缀缓存下当前步命中 ≈ 它)+ 后端是否报过 cache 命中
|
|
194
|
+
// (从不报 cache 的后端不估算,避免虚显 ↻)。
|
|
195
|
+
let lastStepPromptTokens = 0;
|
|
196
|
+
let providerCacheSeen = false;
|
|
329
197
|
const addUsage = (u) => {
|
|
330
198
|
if (!u)
|
|
331
199
|
return;
|
|
@@ -340,48 +208,30 @@ export async function runAgentCore(opts) {
|
|
|
340
208
|
: u;
|
|
341
209
|
};
|
|
342
210
|
const addToolUsage = (outcome) => addUsage(outcome.usage);
|
|
343
|
-
// Only consecutive identical failures are thrashing. Any success (including
|
|
344
|
-
// a mutation between two reads) or different call resets the streak.
|
|
345
|
-
const recordAndHint = createThrashTracker();
|
|
346
211
|
history.push({ role: 'user', content: userInput });
|
|
347
212
|
// 中断回滚快照:push 用户消息后整段浅拷贝。abort 时 length=0;push(...saved) 还原。
|
|
348
213
|
// 这样中断时至少保留用户消息(及之前的历史);每步工具全部执行完毕后刷新快照,
|
|
349
214
|
// 保留已完成的 assistant+tool_calls+tool 结果,只丢弃当前未完成步骤的消息。
|
|
350
215
|
// 用 slice() 而非 length:maybeCompact 会原地重建(length=0;push(...rebuilt)),savedLen 会失效。
|
|
351
216
|
let savedHistory = history.slice();
|
|
352
|
-
//
|
|
353
|
-
//
|
|
354
|
-
// 子 agent 也在自己的 history 上操作(子 agent 独立 history),文件回滚事务则与主轮共享。
|
|
355
|
-
const dropContext = (filter) => dropContextFromHistory(history, filter);
|
|
356
|
-
// 相关性裁剪 pruner:每个 runAgentCore 实例一个,纯静态、不调 LLM、自动判定 read_file 失效。
|
|
357
|
-
// 开关关闭时为 null,所有 pushToolResult 调用走无 pruner 路径(零行为变化)。
|
|
217
|
+
// Relevance and lifecycle collect provenance during normal work. Neither path
|
|
218
|
+
// rewrites history; exact supersession is applied only by the pressure scheduler.
|
|
358
219
|
const relprune = config.contextRelprune ? createRelevancePruner() : null;
|
|
359
|
-
// 观察者生命周期引擎:每个 runAgentCore 实例一个,纯静态、自动维护 grep/glob/web_search/web_fetch 等
|
|
360
|
-
// producer 与 read/edit/write 的 consumer 引用关系;孤立+老化的非观察类工具自动 STUB。
|
|
361
|
-
// 开关关闭时为 null,所有 pushToolResult / mutation 调用走无 lifecycle 路径(零行为变化)。
|
|
362
|
-
// 引擎需要从已有会话 history 恢复观察结果的年龄和 path 索引;不能只追踪本次
|
|
363
|
-
// runAgentCore,否则跨用户轮次的 grep/glob 永远不会衰减。
|
|
364
220
|
let lifecycle = config.contextLifecycle
|
|
365
221
|
? createLifecycleEngine(history)
|
|
366
222
|
: null;
|
|
367
223
|
runtimeContextState.lifecycleStats = lifecycle?.stats();
|
|
368
224
|
rehydrateArtifacts(runtimeContextState, history);
|
|
369
|
-
//
|
|
370
|
-
//
|
|
225
|
+
// The scheduler is the sole automatic history-rewrite entry point. It runs
|
|
226
|
+
// superseded → stale artifact → old logs/search → compact at real pressure.
|
|
227
|
+
// contextBudget=false keeps only the infrastructure compact fallback.
|
|
371
228
|
const scheduler = config.contextBudget !== false
|
|
372
|
-
? createBudgetScheduler(runtimeContextState)
|
|
229
|
+
? createBudgetScheduler(runtimeContextState)
|
|
373
230
|
: null;
|
|
374
|
-
// age-aware encoder 与 history 数组同寿命;每轮重建一次以覆盖外部 /compact 等原地修改。
|
|
375
|
-
const ageAware = config.contextOptimize ? ageAwareStateFor(history) : null;
|
|
376
|
-
ageAware?.rehydrate(history);
|
|
377
231
|
// 本轮流式状态:首个正文 token 到达即停 spinner(思考期间 spinner 持续转「思考中…」,不写思考内容)。
|
|
378
232
|
let mode = 'idle';
|
|
379
233
|
let gotText = false;
|
|
380
234
|
let lastChar = '';
|
|
381
|
-
// 早退重探:本 turn 已执行过工具但模型突然返回无工具调用 + 空文本 → 推一条提示让模型继续。
|
|
382
|
-
// 短回复也可能是合法完成结果,不再按字符数误判;每 turn 最多触发 1 次以防死循环。
|
|
383
|
-
let nudgeCount = 0;
|
|
384
|
-
let hadToolsThisTurn = false;
|
|
385
235
|
const onText = (s) => {
|
|
386
236
|
hooks.onText?.(s); // 主 agent:走 markdown 渲染写内容区
|
|
387
237
|
mode = 'text';
|
|
@@ -408,7 +258,6 @@ export async function runAgentCore(opts) {
|
|
|
408
258
|
hooks.onAbort?.();
|
|
409
259
|
history.length = 0;
|
|
410
260
|
history.push(...savedHistory);
|
|
411
|
-
ageAware?.rehydrate(history);
|
|
412
261
|
setAgentMode(savedMode);
|
|
413
262
|
};
|
|
414
263
|
try {
|
|
@@ -427,7 +276,6 @@ export async function runAgentCore(opts) {
|
|
|
427
276
|
terminationReason: 'aborted',
|
|
428
277
|
finalText: null,
|
|
429
278
|
usage: turnUsage,
|
|
430
|
-
validation: latestValidation,
|
|
431
279
|
changedFiles: mutation.changedFiles.map((item) => item.path),
|
|
432
280
|
};
|
|
433
281
|
}
|
|
@@ -439,10 +287,9 @@ export async function runAgentCore(opts) {
|
|
|
439
287
|
const storedCalibration = getTokenCalibration(requestBaseURL, requestModel, activeTools);
|
|
440
288
|
runtimeContextState.correction = storedCalibration.correction;
|
|
441
289
|
runtimeContextState.calibrationSamples = storedCalibration.samples;
|
|
442
|
-
//
|
|
443
|
-
//
|
|
444
|
-
|
|
445
|
-
// 步前:五区 Budget Scheduler 在优化后的 history 上决策;开关关闭时退化回原 maybeCompact 路径。
|
|
290
|
+
// The scheduler is the only automatic path that may compress old evidence.
|
|
291
|
+
// Normal tool pushes and lifecycle tracking remain metadata-only.
|
|
292
|
+
// 步前:五区 Budget Scheduler 在当前完整 history 上决策;开关关闭时退化回 maybeCompact 路径。
|
|
446
293
|
// 此时 spinner 已停,通知行干净。
|
|
447
294
|
let historyRebuilt = false;
|
|
448
295
|
const compactStartedAt = Date.now();
|
|
@@ -478,8 +325,9 @@ export async function runAgentCore(opts) {
|
|
|
478
325
|
lifecycle = createLifecycleEngine(history);
|
|
479
326
|
runtimeContextState.lifecycleStats = lifecycle.stats();
|
|
480
327
|
}
|
|
481
|
-
ageAware?.rehydrate(history);
|
|
482
328
|
rehydrateArtifacts(runtimeContextState, history);
|
|
329
|
+
// ② compact 后把活跃 plan 重注入系统提示,避免 agent 因上下文压缩丢失执行计划。
|
|
330
|
+
reinjectActivePlanIntoSystem(history);
|
|
483
331
|
}
|
|
484
332
|
hooks.onStepStart?.(); // 主 agent:spinner.start('思考中')
|
|
485
333
|
mode = 'idle';
|
|
@@ -489,12 +337,9 @@ export async function runAgentCore(opts) {
|
|
|
489
337
|
const modelStartedAt = Date.now();
|
|
490
338
|
const provider = safeProviderId(requestBaseURL);
|
|
491
339
|
emitTrace('model_start', { model: requestModel, provider });
|
|
492
|
-
const dynamicSystemSuffix =
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
? '## Post-compaction recovery\nContext was compacted before this request. Re-establish the current objective and unresolved work from retained evidence or the session note, avoid repeating completed investigation, and re-read exact file context before any dependent edit.'
|
|
496
|
-
: '',
|
|
497
|
-
].filter(Boolean).join('\n\n');
|
|
340
|
+
const dynamicSystemSuffix = historyRebuilt
|
|
341
|
+
? '## Post-compaction recovery\nContext was compacted before this request. Re-establish the current objective and unresolved work from retained evidence or the session note, avoid repeating completed investigation, and re-read exact file context before any dependent edit.'
|
|
342
|
+
: '';
|
|
498
343
|
const systemMessage = history[0];
|
|
499
344
|
const requestHistory = dynamicSystemSuffix
|
|
500
345
|
&& systemMessage?.role === 'system'
|
|
@@ -507,10 +352,32 @@ export async function runAgentCore(opts) {
|
|
|
507
352
|
...history.slice(1),
|
|
508
353
|
]
|
|
509
354
|
: history;
|
|
355
|
+
// 实时用量:当前步 prompt 估算(含校准系数)+ 流式累计 completion 估算,
|
|
356
|
+
// 叠上已完成步的实测 turnUsage,经 onLiveUsage 推给底栏实时 chip。
|
|
357
|
+
// turnUsage 在闭包里被 addUsage 原地更新,reportLive 每次调用读最新值。
|
|
358
|
+
const stepPromptEst = estimatePromptTokens(requestHistory, activeTools, runtimeContextState.correction);
|
|
359
|
+
const reportLive = (p) => {
|
|
360
|
+
// 当前步 prompt:末尾 usage chunk 到达后用实测,流式期间用估算(含校准)。
|
|
361
|
+
// 当前步 cache 命中同理:上报即用实测;流式期间按前缀缓存估算 ≈ 上一步实测 prompt
|
|
362
|
+
// (当前 prompt 总含其为前缀),不超过当前步 prompt;后端从不报 cache 时不估算。
|
|
363
|
+
// 口径与轮末摘要一致:chip ↑ 显计费 prompt(裸 - cached),↓/↻ 同。
|
|
364
|
+
const curPrompt = p.promptTokens ?? stepPromptEst;
|
|
365
|
+
const curCached = p.cachedTokens ?? (providerCacheSeen
|
|
366
|
+
? Math.min(lastStepPromptTokens, curPrompt)
|
|
367
|
+
: 0);
|
|
368
|
+
hooks.onLiveUsage?.({
|
|
369
|
+
promptTokens: (turnUsage?.promptTokens ?? 0) + curPrompt,
|
|
370
|
+
completionTokens: (turnUsage?.completionTokens ?? 0) + p.completionTokens,
|
|
371
|
+
totalTokens: (turnUsage?.totalTokens ?? 0) + curPrompt + p.completionTokens,
|
|
372
|
+
cachedTokens: (turnUsage?.cachedTokens ?? 0) + curCached,
|
|
373
|
+
});
|
|
374
|
+
};
|
|
375
|
+
reportLive({ completionTokens: 0 }); // 思考阶段先显 ↑ prompt 估算,首 token 到达后 ↓ 开始涨
|
|
510
376
|
try {
|
|
511
377
|
result = await chat(requestHistory, {
|
|
512
378
|
onText,
|
|
513
379
|
onToolCall,
|
|
380
|
+
onProgress: reportLive,
|
|
514
381
|
onRetry: (retry) => emitTrace('model_retry', {
|
|
515
382
|
model: requestModel,
|
|
516
383
|
provider,
|
|
@@ -547,7 +414,6 @@ export async function runAgentCore(opts) {
|
|
|
547
414
|
terminationReason: 'aborted',
|
|
548
415
|
finalText: null,
|
|
549
416
|
usage: turnUsage,
|
|
550
|
-
validation: latestValidation,
|
|
551
417
|
changedFiles: mutation.changedFiles.map((item) => item.path),
|
|
552
418
|
};
|
|
553
419
|
}
|
|
@@ -566,6 +432,11 @@ export async function runAgentCore(opts) {
|
|
|
566
432
|
});
|
|
567
433
|
runtimeContextState.lastUsage = result.usage; // 供 /context 与状态行显示实测 token
|
|
568
434
|
addUsage(result.usage); // 本轮累计:onDone 摘要行 + AgentRunResult.usage 透传
|
|
435
|
+
if (result.usage) {
|
|
436
|
+
lastStepPromptTokens = result.usage.promptTokens; // 下一步流式期 ↻ 估算的前缀基准
|
|
437
|
+
if (result.usage.cachedTokens > 0)
|
|
438
|
+
providerCacheSeen = true;
|
|
439
|
+
}
|
|
569
440
|
// 用本次实际发送的 tools 计算分母,再以 EWMA 更新 provider/model/tool-set 校准。
|
|
570
441
|
// 只持久化比例与样本数;无 usage 或短 prompt 时保持既有值。
|
|
571
442
|
if (result.usage?.promptTokens && result.usage.promptTokens > 100) {
|
|
@@ -579,7 +450,6 @@ export async function runAgentCore(opts) {
|
|
|
579
450
|
onContextUpdate?.();
|
|
580
451
|
if (result.toolCalls.length > 0) {
|
|
581
452
|
toolCallCount += result.toolCalls.length;
|
|
582
|
-
hadToolsThisTurn = true;
|
|
583
453
|
// 流式正文末尾补换行(若 onToolCall 已补则 lastChar='\n',此处 no-op);防 ● 行黏在正文行尾
|
|
584
454
|
if (mode !== 'idle' && lastChar !== '\n')
|
|
585
455
|
hooks.onTextEnd?.();
|
|
@@ -593,21 +463,19 @@ export async function runAgentCore(opts) {
|
|
|
593
463
|
function: { name: tc.name, arguments: tc.arguments },
|
|
594
464
|
})),
|
|
595
465
|
});
|
|
596
|
-
//
|
|
597
|
-
//
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
//
|
|
601
|
-
|
|
466
|
+
// A+B:记录本步第一条 tool_result 的下标 + 执行前 notes.md 的 mtime,
|
|
467
|
+
// 工具全部执行完后据此判断"本步是否改动了 notes.md"(重同步 / nag)。
|
|
468
|
+
const toolResultStartIdx = history.length;
|
|
469
|
+
const notesMtimeBefore = getNotesMtime();
|
|
470
|
+
// Record interstitial narration for observability only. It never changes tool output
|
|
471
|
+
// or injects instructions back into the model context.
|
|
472
|
+
const narration = result.content?.trim() ?? '';
|
|
602
473
|
if (narration) {
|
|
603
474
|
emitTrace('narration', {
|
|
604
|
-
chars: narration.
|
|
475
|
+
chars: [...narration].length,
|
|
605
476
|
toolCalls: result.toolCalls.length,
|
|
606
|
-
budget: NARRATION_CHAR_BUDGET,
|
|
607
|
-
overBudget: narration.overBudget,
|
|
608
477
|
step,
|
|
609
478
|
});
|
|
610
|
-
pendingNarrationHint = narration.hint;
|
|
611
479
|
}
|
|
612
480
|
// 工具分组执行(保 tool_calls 原顺序):safe parallel 工具照常并发;连续
|
|
613
481
|
// resource-locked mutation 先按序完成权限预检,再按 canonical resource lock 启动。
|
|
@@ -615,6 +483,7 @@ export async function runAgentCore(opts) {
|
|
|
615
483
|
// 串行工具仍是本调用列表内的屏障;渲染/history 回灌始终按原 tool_calls 顺序。
|
|
616
484
|
// executeToolOutcome 永不抛错,失败通过结构化 status/code 返回。
|
|
617
485
|
const calls = result.toolCalls;
|
|
486
|
+
const modelAttachments = [];
|
|
618
487
|
const tracedCalls = calls.map((tc, index) => ({
|
|
619
488
|
toolCallId: `${traceTurnId}:step:${step}:tool:${index}`,
|
|
620
489
|
args: summarizeToolArguments(tc.arguments),
|
|
@@ -626,28 +495,15 @@ export async function runAgentCore(opts) {
|
|
|
626
495
|
tool: tc.name,
|
|
627
496
|
argumentHash: traceCall.args.sha256,
|
|
628
497
|
arguments: traceCall.args,
|
|
629
|
-
attempt: 1,
|
|
630
|
-
retry: 0,
|
|
631
498
|
}, {
|
|
632
499
|
toolCallId: traceCall.toolCallId,
|
|
633
500
|
...(tc.id ? { providerToolCallId: tc.id } : {}),
|
|
634
501
|
});
|
|
635
502
|
}
|
|
636
|
-
const traceToolRetry = (tc, index, retry) => {
|
|
637
|
-
const traceCall = tracedCalls[index];
|
|
638
|
-
emitTrace('tool_retry', {
|
|
639
|
-
tool: tc.name,
|
|
640
|
-
argumentHash: traceCall.args.sha256,
|
|
641
|
-
attempt: retry.attempt,
|
|
642
|
-
nextAttempt: retry.nextAttempt,
|
|
643
|
-
waitMs: retry.waitMs,
|
|
644
|
-
code: retry.code,
|
|
645
|
-
}, {
|
|
646
|
-
toolCallId: traceCall.toolCallId,
|
|
647
|
-
...(tc.id ? { providerToolCallId: tc.id } : {}),
|
|
648
|
-
});
|
|
649
|
-
};
|
|
650
503
|
const traceToolEnd = (tc, index, outcome) => {
|
|
504
|
+
if (outcome.status === 'success' && outcome.modelAttachments?.length) {
|
|
505
|
+
modelAttachments.push(...outcome.modelAttachments);
|
|
506
|
+
}
|
|
651
507
|
const traceCall = tracedCalls[index];
|
|
652
508
|
emitTrace('tool_call_end', {
|
|
653
509
|
tool: tc.name,
|
|
@@ -656,9 +512,6 @@ export async function runAgentCore(opts) {
|
|
|
656
512
|
code: outcome.code,
|
|
657
513
|
retryable: outcome.retryable,
|
|
658
514
|
durationMs: outcome.durationMs ?? 0,
|
|
659
|
-
attempt: outcome.attempts ?? 1,
|
|
660
|
-
retry: Math.max(0, (outcome.attempts ?? 1) - 1),
|
|
661
|
-
retryDelayMs: outcome.retryDelayMs ?? 0,
|
|
662
515
|
changedFiles: outcome.changedFiles ?? [],
|
|
663
516
|
staleFiles: outcome.staleFiles ?? [],
|
|
664
517
|
...(outcome.changeSet ? { changeSet: outcome.changeSet } : {}),
|
|
@@ -683,8 +536,7 @@ export async function runAgentCore(opts) {
|
|
|
683
536
|
durationMs: 0,
|
|
684
537
|
};
|
|
685
538
|
hooks.onToolResult?.(currentCall, error, null, null, 1);
|
|
686
|
-
|
|
687
|
-
pushToolResult(history, currentCall, hint ? `${error}${hint}` : error, relprune, lifecycle, scheduler, runtimeContextState, false);
|
|
539
|
+
pushToolResult(history, currentCall, error, relprune, lifecycle, scheduler, runtimeContextState, false);
|
|
688
540
|
traceToolEnd(currentCall, i, outcome);
|
|
689
541
|
i++;
|
|
690
542
|
continue;
|
|
@@ -702,10 +554,7 @@ export async function runAgentCore(opts) {
|
|
|
702
554
|
for (const tc of batch)
|
|
703
555
|
hooks.onToolHeader?.(tc);
|
|
704
556
|
hooks.onToolStart?.(batch[0].name);
|
|
705
|
-
const started = batch.map((tc
|
|
706
|
-
dropContext,
|
|
707
|
-
onRetry: (retry) => traceToolRetry(tc, i + offset, retry),
|
|
708
|
-
}));
|
|
557
|
+
const started = batch.map((tc) => executeToolOutcome(tc.name, tc.arguments, signal));
|
|
709
558
|
for (let k = 0; k < batch.length; k++) {
|
|
710
559
|
const tc = batch[k];
|
|
711
560
|
const outcome = await started[k];
|
|
@@ -714,34 +563,15 @@ export async function runAgentCore(opts) {
|
|
|
714
563
|
traceToolEnd(tc, i + k, outcome);
|
|
715
564
|
const output = outcome.output;
|
|
716
565
|
hooks.onToolResult?.(tc, output, null, null, 1); // 并行工具无 diff
|
|
717
|
-
// Thrashing:history 里附 hint(UI 已用干净 output 渲染,避免屏幕噪声)
|
|
718
|
-
const hint = recordAndHint(tc.name, tc.arguments, outcome.status === 'success');
|
|
719
|
-
const { output: annotated, reflectionCategory } = appendRetryAnnotations(output, outcome.status, outcome.code, hint);
|
|
720
|
-
// QUAL-01 trace 硬事件:反思指针注入计数(history 文本扫描会因 LLM 在
|
|
721
|
-
// 正文里提一句 "[retry reflection]" 而误算,改用 core 接缝处显式 emit)。
|
|
722
|
-
if (reflectionCategory !== null) {
|
|
723
|
-
emitTrace('retry_reflection', {
|
|
724
|
-
tool: tc.name,
|
|
725
|
-
code: outcome.code,
|
|
726
|
-
category: reflectionCategory,
|
|
727
|
-
attempt: outcome.attempts ?? 1,
|
|
728
|
-
}, tc.id ? { providerToolCallId: tc.id } : {});
|
|
729
|
-
}
|
|
730
|
-
// ASK-01: 计数 + 预算提示
|
|
731
566
|
if (tc.name === 'ask_human' && outcome.status === 'success') {
|
|
732
567
|
askHumanCountThisTurn += 1;
|
|
733
|
-
// QUAL-01 trace 硬事件:ask_human 真实调用计数(取代 tool_call_end.name 推断)。
|
|
734
568
|
emitTrace('ask_human_call', {
|
|
735
569
|
tool: tc.name,
|
|
736
570
|
status: outcome.status,
|
|
737
571
|
perTurnCount: askHumanCountThisTurn,
|
|
738
572
|
}, tc.id ? { providerToolCallId: tc.id } : {});
|
|
739
573
|
}
|
|
740
|
-
|
|
741
|
-
// NARR-01: 旁白回压提示只挂本批第一条结果(takeNarrationHint 自清空)。
|
|
742
|
-
const narrationHint = takeNarrationHint();
|
|
743
|
-
const finalOutput = `${annotated}${askBudget ?? ''}${narrationHint ?? ''}`;
|
|
744
|
-
pushToolResult(history, tc, finalOutput, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
574
|
+
pushToolResult(history, tc, output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
745
575
|
}
|
|
746
576
|
hooks.onToolDone?.();
|
|
747
577
|
i = j;
|
|
@@ -794,14 +624,12 @@ export async function runAgentCore(opts) {
|
|
|
794
624
|
const firstAllowed = entries.find((entry) => !entry.denied);
|
|
795
625
|
if (firstAllowed)
|
|
796
626
|
hooks.onToolStart?.(firstAllowed.tc.name);
|
|
797
|
-
const started = entries.map((entry
|
|
627
|
+
const started = entries.map((entry) => entry.denied
|
|
798
628
|
? Promise.resolve(entry.denied)
|
|
799
629
|
: executeToolOutcome(entry.tc.name, entry.tc.arguments, signal, {
|
|
800
|
-
dropContext,
|
|
801
630
|
onLockAcquired: (lockedArgs) => {
|
|
802
631
|
entry.diff = readDiffContext(entry.tc, lockedArgs);
|
|
803
632
|
},
|
|
804
|
-
onRetry: (retry) => traceToolRetry(entry.tc, i + offset, retry),
|
|
805
633
|
}));
|
|
806
634
|
for (let k = 0; k < entries.length; k++) {
|
|
807
635
|
const entry = entries[k];
|
|
@@ -810,8 +638,7 @@ export async function runAgentCore(opts) {
|
|
|
810
638
|
opts.onToolOutcome?.(entry.tc.name, entry.parsed ?? {}, outcome);
|
|
811
639
|
traceToolEnd(entry.tc, i + k, outcome);
|
|
812
640
|
hooks.onToolResult?.(entry.tc, outcome.output, entry.denied ? null : entry.parsed, entry.diff.preWriteOld, entry.diff.editStartLine);
|
|
813
|
-
|
|
814
|
-
pushToolResult(history, entry.tc, hint ? `${outcome.output}${hint}` : outcome.output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
641
|
+
pushToolResult(history, entry.tc, outcome.output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
815
642
|
const invalidatedFiles = [...new Set([
|
|
816
643
|
...(outcome.changedFiles ?? []),
|
|
817
644
|
...(outcome.staleFiles ?? []),
|
|
@@ -846,9 +673,7 @@ export async function runAgentCore(opts) {
|
|
|
846
673
|
durationMs: 0,
|
|
847
674
|
};
|
|
848
675
|
hooks.onToolResult?.(tc, err, null, null, 1);
|
|
849
|
-
|
|
850
|
-
const hint = recordAndHint(tc.name, tc.arguments, false);
|
|
851
|
-
pushToolResult(history, tc, hint ? `${err}${hint}` : err, relprune, lifecycle, scheduler);
|
|
676
|
+
pushToolResult(history, tc, err, relprune, lifecycle, scheduler);
|
|
852
677
|
traceToolEnd(tc, i, outcome);
|
|
853
678
|
i++;
|
|
854
679
|
continue;
|
|
@@ -877,8 +702,7 @@ export async function runAgentCore(opts) {
|
|
|
877
702
|
hooks.onToolHeader?.(tc);
|
|
878
703
|
const outcome = deniedOutcome(tc.name);
|
|
879
704
|
hooks.onToolResult?.(tc, outcome.output, null, null, 1);
|
|
880
|
-
|
|
881
|
-
pushToolResult(history, tc, hint ? `${outcome.output}${hint}` : outcome.output, relprune, lifecycle, scheduler, runtimeContextState, false);
|
|
705
|
+
pushToolResult(history, tc, outcome.output, relprune, lifecycle, scheduler, runtimeContextState, false);
|
|
882
706
|
traceToolEnd(tc, i, outcome);
|
|
883
707
|
i++;
|
|
884
708
|
continue;
|
|
@@ -891,12 +715,10 @@ export async function runAgentCore(opts) {
|
|
|
891
715
|
let diff = readDiffContext(tc, mutationParsed);
|
|
892
716
|
hooks.onToolStart?.(tc.name);
|
|
893
717
|
const outcome = await executeToolOutcome(tc.name, tc.arguments, signal, {
|
|
894
|
-
dropContext,
|
|
895
718
|
onLockAcquired: (lockedArgs) => {
|
|
896
719
|
if (mutationParsed)
|
|
897
720
|
diff = readDiffContext(tc, lockedArgs);
|
|
898
721
|
},
|
|
899
|
-
onRetry: (retry) => traceToolRetry(tc, i, retry),
|
|
900
722
|
});
|
|
901
723
|
addToolUsage(outcome);
|
|
902
724
|
opts.onToolOutcome?.(tc.name, parsed ?? {}, outcome);
|
|
@@ -904,35 +726,15 @@ export async function runAgentCore(opts) {
|
|
|
904
726
|
const output = outcome.output;
|
|
905
727
|
hooks.onToolDone?.();
|
|
906
728
|
hooks.onToolResult?.(tc, output, mutationParsed, diff.preWriteOld, diff.editStartLine);
|
|
907
|
-
// Thrashing:同上(history 附 hint,UI 干净)
|
|
908
|
-
const hint = recordAndHint(tc.name, tc.arguments, outcome.status === 'success');
|
|
909
|
-
// RETRY-01: 同 parallel 路径,在非成功状态追加反思指针。
|
|
910
|
-
const { output: annotated, reflectionCategory } = appendRetryAnnotations(output, outcome.status, outcome.code, hint);
|
|
911
|
-
// QUAL-01 trace 硬事件:反思指针注入计数(history 文本扫描会因 LLM 在
|
|
912
|
-
// 正文里提一句 "[retry reflection]" 而误算,改用 core 接缝处显式 emit)。
|
|
913
|
-
if (reflectionCategory !== null) {
|
|
914
|
-
emitTrace('retry_reflection', {
|
|
915
|
-
tool: tc.name,
|
|
916
|
-
code: outcome.code,
|
|
917
|
-
category: reflectionCategory,
|
|
918
|
-
attempt: outcome.attempts ?? 1,
|
|
919
|
-
}, tc.id ? { providerToolCallId: tc.id } : {});
|
|
920
|
-
}
|
|
921
|
-
// ASK-01: ask_human 计数 + 预算提示(同 parallel 路径)。
|
|
922
729
|
if (tc.name === 'ask_human' && outcome.status === 'success') {
|
|
923
730
|
askHumanCountThisTurn += 1;
|
|
924
|
-
// QUAL-01 trace 硬事件:ask_human 真实调用计数(取代 tool_call_end.name 推断)。
|
|
925
731
|
emitTrace('ask_human_call', {
|
|
926
732
|
tool: tc.name,
|
|
927
733
|
status: outcome.status,
|
|
928
734
|
perTurnCount: askHumanCountThisTurn,
|
|
929
735
|
}, tc.id ? { providerToolCallId: tc.id } : {});
|
|
930
736
|
}
|
|
931
|
-
|
|
932
|
-
// NARR-01: 旁白回压提示只挂本批第一条结果(takeNarrationHint 自清空)。
|
|
933
|
-
const narrationHint = takeNarrationHint();
|
|
934
|
-
const finalOutput = `${annotated}${askBudget ?? ''}${narrationHint ?? ''}`;
|
|
935
|
-
pushToolResult(history, tc, finalOutput, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
737
|
+
pushToolResult(history, tc, output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
|
|
936
738
|
const invalidatedFiles = [...new Set([
|
|
937
739
|
...(outcome.changedFiles ?? []),
|
|
938
740
|
...(outcome.staleFiles ?? []),
|
|
@@ -948,6 +750,48 @@ export async function runAgentCore(opts) {
|
|
|
948
750
|
i++;
|
|
949
751
|
}
|
|
950
752
|
}
|
|
753
|
+
// A(事件驱动重同步):本步若改动了 notes.md,把最新 plan 块刷回 history[0],
|
|
754
|
+
// 让模型上下文镜像当前勾选态(不再停留在轮首的旧副本)。只在 mtime 变化时触发,零额外 churn。
|
|
755
|
+
// B(nag 提醒):连续 N 步有工具活动但没更新 plan,在当前步第一条 tool_result 前注入提醒。
|
|
756
|
+
const notesMtimeAfter = getNotesMtime();
|
|
757
|
+
if (notesMtimeAfter !== notesMtimeBefore) {
|
|
758
|
+
reinjectActivePlanIntoSystem(history);
|
|
759
|
+
stepsSincePlanTouch = 0;
|
|
760
|
+
}
|
|
761
|
+
else {
|
|
762
|
+
stepsSincePlanTouch += 1;
|
|
763
|
+
if (stepsSincePlanTouch >= PLAN_NAG_THRESHOLD) {
|
|
764
|
+
const activePlan = extractActivePlanSection();
|
|
765
|
+
const firstToolMsg = history[toolResultStartIdx];
|
|
766
|
+
if (activePlan && firstToolMsg && firstToolMsg.role === 'tool' && typeof firstToolMsg.content === 'string') {
|
|
767
|
+
firstToolMsg.content = `${PLAN_NAG_TEXT}\n\n${firstToolMsg.content}`;
|
|
768
|
+
}
|
|
769
|
+
stepsSincePlanTouch = 0;
|
|
770
|
+
}
|
|
771
|
+
}
|
|
772
|
+
if (modelAttachments.length > 0) {
|
|
773
|
+
const names = modelAttachments.map((attachment) => attachment.name).join(', ');
|
|
774
|
+
const content = [
|
|
775
|
+
{
|
|
776
|
+
type: 'text',
|
|
777
|
+
text: `The view_image tool loaded the following visual input: ${names}. Analyze the attached image content directly.`,
|
|
778
|
+
},
|
|
779
|
+
...modelAttachments.map((attachment) => ({
|
|
780
|
+
type: 'image_url',
|
|
781
|
+
image_url: {
|
|
782
|
+
url: attachment.dataUrl,
|
|
783
|
+
// 不默认补 auto: OpenAI 省略时等同 auto,但 MiniMax 仅接受 low/default/high。
|
|
784
|
+
// 让各 provider 采用默认枚举;仅保留工具明确请求的 low/high。
|
|
785
|
+
...(attachment.detail === 'low' || attachment.detail === 'high'
|
|
786
|
+
? { detail: attachment.detail }
|
|
787
|
+
: {}),
|
|
788
|
+
},
|
|
789
|
+
})),
|
|
790
|
+
];
|
|
791
|
+
// OpenAI tool-call protocol requires every tool result to immediately follow the
|
|
792
|
+
// assistant tool_calls message; append visual input only after the full batch.
|
|
793
|
+
history.push({ role: 'user', content });
|
|
794
|
+
}
|
|
951
795
|
// 工具步末尾补一空行:与下一轮的思考 / 正文分隔(否则 ↳ 后紧接 ▎ 思考,无空行不好看;
|
|
952
796
|
// 与正文→● 的 1 空行对称)。工具结果已以 \n 收尾,此处再补 \n 恰好 1 空行。
|
|
953
797
|
hooks.onToolBatchEnd?.();
|
|
@@ -958,169 +802,20 @@ export async function runAgentCore(opts) {
|
|
|
958
802
|
}
|
|
959
803
|
if (mode !== 'idle' && lastChar !== '\n')
|
|
960
804
|
hooks.onTextEnd?.(); // 流式末尾补换行
|
|
961
|
-
//
|
|
962
|
-
//
|
|
963
|
-
// 早退重探只处理真正的空回复。短回复可能是合法完成结果(例如精确状态标记);
|
|
964
|
-
// 仅凭字符数继续调用会重复输出,并产生一次完整的额外模型请求。
|
|
965
|
-
const replyIsEmpty = (result.content?.trim().length ?? 0) === 0;
|
|
966
|
-
if (hadToolsThisTurn && nudgeCount < 1 && replyIsEmpty) {
|
|
967
|
-
nudgeCount++;
|
|
968
|
-
history.push({ role: 'assistant', content: result.content });
|
|
969
|
-
history.push({
|
|
970
|
-
role: 'user',
|
|
971
|
-
content: 'You stopped before completing the task. Please continue investigating — call more tools if needed, or provide a complete answer based on what you have gathered so far.',
|
|
972
|
-
});
|
|
973
|
-
continue; // 带着提示再调一次 LLM
|
|
974
|
-
}
|
|
975
|
-
// 没有工具调用:候选正文已流式打印。若本轮有新的代码变更,先通过框架验证门;
|
|
976
|
-
// failed 作为 system observation 风格的 user 消息回灌,不能伪造无配对的 tool 消息。
|
|
805
|
+
// 没有工具调用:接受 agent 的完成判断。框架不自动运行测试、构建或完成门,
|
|
806
|
+
// 也不因缺少验证证据强制追加模型轮次;agent 仍可自行调用工具验证。
|
|
977
807
|
if (!gotText)
|
|
978
808
|
hooks.onNoReply?.();
|
|
979
|
-
|
|
980
|
-
const mutationBeforeValidation = getCurrentTurnMutationState();
|
|
981
|
-
const shouldValidate = opts.autoValidate === true &&
|
|
982
|
-
getAgentMode() !== 'plan' &&
|
|
983
|
-
(mutationBeforeValidation.version > validatedMutationVersion || latestValidation?.status === 'failed');
|
|
984
|
-
if (shouldValidate) {
|
|
985
|
-
history.push(candidate);
|
|
986
|
-
emitTrace('validation_start', {
|
|
987
|
-
mutationVersion: mutationBeforeValidation.version,
|
|
988
|
-
changedFiles: mutationBeforeValidation.changedFiles.map((item) => item.path),
|
|
989
|
-
});
|
|
990
|
-
try {
|
|
991
|
-
latestValidation = await validator(signal, {
|
|
992
|
-
onCommandStart: (command) => hooks.onValidationStart?.(command),
|
|
993
|
-
onPermissionDecision: (permission) => {
|
|
994
|
-
const summary = summarizeToolArguments(JSON.stringify(permission.arguments));
|
|
995
|
-
emitTrace('permission', {
|
|
996
|
-
source: 'automatic_validation',
|
|
997
|
-
tool: permission.tool,
|
|
998
|
-
decision: permission.decision,
|
|
999
|
-
argumentHash: summary.sha256,
|
|
1000
|
-
});
|
|
1001
|
-
},
|
|
1002
|
-
});
|
|
1003
|
-
}
|
|
1004
|
-
catch (error) {
|
|
1005
|
-
const mutation = getCurrentTurnMutationState();
|
|
1006
|
-
const message = `Automatic validation failed to run: ${error instanceof Error ? error.message : String(error)}`;
|
|
1007
|
-
const status = signal?.aborted ? 'aborted' : 'failed';
|
|
1008
|
-
latestValidation = {
|
|
1009
|
-
status,
|
|
1010
|
-
level: 'V0',
|
|
1011
|
-
output: message,
|
|
1012
|
-
durationMs: 0,
|
|
1013
|
-
diagnostics: [{
|
|
1014
|
-
level: 'V0', source: 'verifier', severity: 'error', code: 'VERIFIER_ERROR', message,
|
|
1015
|
-
}],
|
|
1016
|
-
stages: [],
|
|
1017
|
-
verificationComplete: false,
|
|
1018
|
-
fingerprint: `verifier-error-${mutation.version}`,
|
|
1019
|
-
inputFingerprint: `mutation-${mutation.version}`,
|
|
1020
|
-
inputMutationVersion: mutation.version,
|
|
1021
|
-
affectedPackages: [],
|
|
1022
|
-
changedFiles: mutation.changedFiles.map((item) => item.path),
|
|
1023
|
-
mutationVersion: mutation.version,
|
|
1024
|
-
};
|
|
1025
|
-
}
|
|
1026
|
-
emitTrace('validation_end', {
|
|
1027
|
-
status: latestValidation.status,
|
|
1028
|
-
level: latestValidation.level,
|
|
1029
|
-
durationMs: latestValidation.durationMs,
|
|
1030
|
-
verificationComplete: latestValidation.verificationComplete,
|
|
1031
|
-
skipReason: latestValidation.skipReason,
|
|
1032
|
-
fingerprint: latestValidation.fingerprint,
|
|
1033
|
-
mutationVersion: latestValidation.mutationVersion,
|
|
1034
|
-
stages: latestValidation.stages.map((stage) => ({
|
|
1035
|
-
level: stage.level,
|
|
1036
|
-
status: stage.status,
|
|
1037
|
-
adapter: stage.adapter,
|
|
1038
|
-
code: stage.diagnostics[0]?.code,
|
|
1039
|
-
durationMs: stage.durationMs,
|
|
1040
|
-
cached: stage.cached === true,
|
|
1041
|
-
})),
|
|
1042
|
-
});
|
|
1043
|
-
hooks.onValidationResult?.(latestValidation);
|
|
1044
|
-
validatedMutationVersion = latestValidation.mutationVersion;
|
|
1045
|
-
if (latestValidation.status === 'passed' && latestValidation.verificationComplete) {
|
|
1046
|
-
fullyValidatedMutationVersion = latestValidation.mutationVersion;
|
|
1047
|
-
}
|
|
1048
|
-
if (signal?.aborted || latestValidation.status === 'aborted') {
|
|
1049
|
-
abortRestore();
|
|
1050
|
-
traceStatus = 'aborted';
|
|
1051
|
-
const mutation = getCurrentTurnMutationState();
|
|
1052
|
-
return {
|
|
1053
|
-
completed: false,
|
|
1054
|
-
terminationReason: 'aborted',
|
|
1055
|
-
finalText: null,
|
|
1056
|
-
usage: turnUsage,
|
|
1057
|
-
validation: latestValidation,
|
|
1058
|
-
changedFiles: mutation.changedFiles.map((item) => item.path),
|
|
1059
|
-
};
|
|
1060
|
-
}
|
|
1061
|
-
if (latestValidation.status === 'failed') {
|
|
1062
|
-
history.push({
|
|
1063
|
-
role: 'user',
|
|
1064
|
-
content: '[System observation: automatic validation failed]\n' +
|
|
1065
|
-
`Command: ${latestValidation.command ?? '(internal verifier)'}\n` +
|
|
1066
|
-
`${latestValidation.output}\n\n` +
|
|
1067
|
-
'Fix the reported problem, then finish the task. Do not claim success until validation passes.',
|
|
1068
|
-
});
|
|
1069
|
-
// 验证及其失败观察均已完整落入 history;下一步中断时可安全保留。
|
|
1070
|
-
savedHistory = history.slice();
|
|
1071
|
-
continue;
|
|
1072
|
-
}
|
|
1073
|
-
}
|
|
1074
|
-
else {
|
|
1075
|
-
// PROMPT-02: 硬关卡 — 收尾路径上,如果本 turn 有 mutation 且未通过
|
|
1076
|
-
// validation(autoValidate=false / 验证已 passed 但 LLM 又宣称完成),
|
|
1077
|
-
// 推一条 [checklist] user 消息,强制模型再走一轮显式确认。
|
|
1078
|
-
// 防死循环:checklist 已推过 2 次仍无新工具调用 → 放行(走原本的 push)。
|
|
1079
|
-
const finalMutationForChecklist = getCurrentTurnMutationState();
|
|
1080
|
-
const checklistCtx = {
|
|
1081
|
-
hadMutation: finalMutationForChecklist.version > 0,
|
|
1082
|
-
lastValidationStatus: (latestValidation?.status === 'failed' || latestValidation?.status === 'passed')
|
|
1083
|
-
? latestValidation.status
|
|
1084
|
-
: 'none',
|
|
1085
|
-
mode: getAgentMode(),
|
|
1086
|
-
modelFamily: inferModelFamily(config.model),
|
|
1087
|
-
};
|
|
1088
|
-
const checklistRetryCount = history.__checklistStreak ?? 0;
|
|
1089
|
-
const shouldFireChecklist = preCompletionChecklistHandler !== null
|
|
1090
|
-
&& preCompletionChecklistHandler(checklistCtx)
|
|
1091
|
-
&& checklistRetryCount < 2;
|
|
1092
|
-
if (shouldFireChecklist) {
|
|
1093
|
-
history.push(candidate);
|
|
1094
|
-
history.push({
|
|
1095
|
-
role: 'user',
|
|
1096
|
-
content: _checklistMiddleware.buildUserMessage(checklistCtx.modelFamily),
|
|
1097
|
-
});
|
|
1098
|
-
history.__checklistStreak = checklistRetryCount + 1;
|
|
1099
|
-
// QUAL-01 trace 硬事件:PROMPT-02 触发计数(取代 history 文本扫描
|
|
1100
|
-
// [checklist] marker — LLM 在正文里提一句 "[checklist]" 也会被误算)。
|
|
1101
|
-
emitTrace('checklist_triggered', {
|
|
1102
|
-
streak: checklistRetryCount + 1,
|
|
1103
|
-
validationStatus: checklistCtx.lastValidationStatus,
|
|
1104
|
-
hadMutation: checklistCtx.hadMutation,
|
|
1105
|
-
modelFamily: checklistCtx.modelFamily ?? 'other',
|
|
1106
|
-
});
|
|
1107
|
-
// 重要: 不置 done,让主循环继续下一轮(模型被迫用工具或写出可验证声明)。
|
|
1108
|
-
continue;
|
|
1109
|
-
}
|
|
1110
|
-
history.push(candidate);
|
|
1111
|
-
}
|
|
809
|
+
history.push({ role: 'assistant', content: result.content });
|
|
1112
810
|
const finalMutation = getCurrentTurnMutationState();
|
|
1113
811
|
done = true;
|
|
1114
812
|
traceStatus = 'completed';
|
|
1115
|
-
const verified = finalMutation.version <= fullyValidatedMutationVersion;
|
|
1116
813
|
return {
|
|
1117
814
|
completed: true,
|
|
1118
|
-
terminationReason:
|
|
815
|
+
terminationReason: 'completed',
|
|
1119
816
|
finalText: result.content,
|
|
1120
817
|
usage: turnUsage,
|
|
1121
|
-
validation: latestValidation,
|
|
1122
818
|
changedFiles: finalMutation.changedFiles.map((item) => item.path),
|
|
1123
|
-
history: history.slice(),
|
|
1124
819
|
};
|
|
1125
820
|
}
|
|
1126
821
|
finally {
|
|
@@ -1134,15 +829,12 @@ export async function runAgentCore(opts) {
|
|
|
1134
829
|
done = true;
|
|
1135
830
|
traceStatus = 'max_steps';
|
|
1136
831
|
const finalMutation = getCurrentTurnMutationState();
|
|
1137
|
-
const hasUnverifiedChanges = finalMutation.version > fullyValidatedMutationVersion;
|
|
1138
832
|
return {
|
|
1139
833
|
completed: false,
|
|
1140
|
-
terminationReason:
|
|
834
|
+
terminationReason: 'max_steps',
|
|
1141
835
|
finalText: null,
|
|
1142
836
|
usage: turnUsage,
|
|
1143
|
-
validation: latestValidation,
|
|
1144
837
|
changedFiles: finalMutation.changedFiles.map((item) => item.path),
|
|
1145
|
-
history: history.slice(),
|
|
1146
838
|
};
|
|
1147
839
|
}
|
|
1148
840
|
finally {
|
|
@@ -1154,7 +846,6 @@ export async function runAgentCore(opts) {
|
|
|
1154
846
|
toolCalls: toolCallCount,
|
|
1155
847
|
changedFiles: finalMutation.changedFiles.map((item) => item.path),
|
|
1156
848
|
totalTokens: turnUsage?.totalTokens,
|
|
1157
|
-
validationStatus: latestValidation?.status,
|
|
1158
849
|
});
|
|
1159
850
|
try {
|
|
1160
851
|
opts.onTrace?.({
|
|
@@ -1166,7 +857,6 @@ export async function runAgentCore(opts) {
|
|
|
1166
857
|
toolCalls: toolCallCount,
|
|
1167
858
|
changedFiles: finalMutation.changedFiles.map((item) => item.path),
|
|
1168
859
|
usage: turnUsage,
|
|
1169
|
-
validation: latestValidation,
|
|
1170
860
|
});
|
|
1171
861
|
}
|
|
1172
862
|
catch {
|