mocode-ai 1.1.6 → 1.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +23 -37
  2. package/README.zh-CN.md +46 -38
  3. package/dist/agent/core.js +106 -395
  4. package/dist/agent/index.js +2 -21
  5. package/dist/agent/spawn.js +3 -5
  6. package/dist/agent/work-discipline.js +28 -82
  7. package/dist/config/index.js +7 -7
  8. package/dist/context/age-aware.js +18 -48
  9. package/dist/context/artifacts.js +19 -17
  10. package/dist/context/budget.js +27 -28
  11. package/dist/context/classifier.js +0 -1
  12. package/dist/context/encoders/index.js +4 -11
  13. package/dist/context/index.js +4 -7
  14. package/dist/context/lifecycle.js +115 -483
  15. package/dist/context/pipeline.js +8 -15
  16. package/dist/context/relevance.js +77 -55
  17. package/dist/host/stdio.js +0 -6
  18. package/dist/i18n/index.js +0 -6
  19. package/dist/index.js +11 -1
  20. package/dist/llm/index.js +77 -5
  21. package/dist/mcp/index.js +0 -1
  22. package/dist/repl/index.js +21 -15
  23. package/dist/runtime/browser-manager.js +299 -0
  24. package/dist/runtime/dev-server-manager.js +354 -0
  25. package/dist/runtime/shutdown.js +26 -0
  26. package/dist/session/compact.js +86 -102
  27. package/dist/session/index.js +0 -1
  28. package/dist/session/scheduler.js +88 -92
  29. package/dist/session/trace-metrics.js +5 -92
  30. package/dist/session/trace.js +1 -10
  31. package/dist/tools/builtins/browser.js +199 -0
  32. package/dist/tools/builtins/dev-server.js +99 -0
  33. package/dist/tools/builtins/index.js +28 -19
  34. package/dist/tools/builtins/screenshot.js +173 -0
  35. package/dist/tools/builtins/view-image.js +49 -0
  36. package/dist/tools/constants.js +3 -0
  37. package/dist/tools/registry.js +5 -29
  38. package/dist/ui/layout.js +28 -5
  39. package/dist/ui/render.js +11 -0
  40. package/package.json +2 -2
  41. package/dist/agent/middleware/checklist.js +0 -59
  42. package/dist/session/drop.d.ts +0 -19
  43. package/dist/session/drop.js +0 -93
  44. package/dist/tools/builtins/drop-context.d.ts +0 -18
  45. package/dist/tools/builtins/drop-context.js +0 -68
  46. package/dist/verification/diagnostics.js +0 -108
  47. package/dist/verification/fingerprint.js +0 -54
  48. package/dist/verification/index.js +0 -333
  49. package/dist/verification/postconditions.js +0 -98
  50. package/dist/verification/targeted-tests.js +0 -96
  51. package/dist/verification/types.js +0 -1
@@ -10,14 +10,11 @@ import { executeToolOutcome, getToolCapabilities, isFileMutationTool, tools, } f
10
10
  import { checkPermission } from '../permissions/index.js';
11
11
  import { validateToolArguments } from '../tools/validation.js';
12
12
  import { getPlanDisabledTools, getRuntimeDisabledTools } from '../tools/constants.js';
13
- import { createPreCompletionChecklistMiddleware, } from './middleware/checklist.js';
14
- import { inferModelFamily } from './work-discipline.js';
15
- import { reflectionHint, classifyError } from './retry-classifier.js';
16
13
  import { getAgentMode, setAgentMode } from './mode.js';
17
- import { maybeCompact, contextState, dropContextFromHistory, createTraceEvent, summarizeToolArguments, safeProviderId, } from '../session/index.js';
14
+ import { maybeCompact, contextState, createTraceEvent, summarizeToolArguments, safeProviderId, } from '../session/index.js';
15
+ import { capToolResultForHistory } from '../session/compact.js';
18
16
  import { createBudgetScheduler } from '../session/scheduler.js';
19
- import { optimizeToolResult, HOT_TURN_WINDOW, userTurnBoundary, recordArtifact, invalidateArtifacts, rehydrateArtifacts, } from '../context/index.js';
20
- import { createAgeAwareEncodingState, } from '../context/age-aware.js';
17
+ import { recordArtifact, invalidateArtifacts, rehydrateArtifacts, } from '../context/index.js';
21
18
  import { createRelevancePruner } from '../context/relevance.js';
22
19
  import { isToolResultSuccess } from '../context/utils.js';
23
20
  import { config } from '../config/index.js';
@@ -26,18 +23,7 @@ import { jailResolve } from '../sandbox/index.js';
26
23
  import { createLifecycleEngine } from '../context/lifecycle.js';
27
24
  import { getTokenCalibration, updateTokenCalibration, } from '../context/token-calibration.js';
28
25
  import { getCurrentTurnId, getCurrentTurnMutationState } from '../rollback/index.js';
29
- import { createAutomaticValidator, } from '../verification/index.js';
30
26
  import { getCurrentSessionId } from '../session/state.js';
31
- /** Stable per-history age state survives user turns; WeakMap avoids retaining closed sessions. */
32
- const ageAwareStateByHistory = new WeakMap();
33
- function ageAwareStateFor(history) {
34
- const existing = ageAwareStateByHistory.get(history);
35
- if (existing)
36
- return existing;
37
- const created = createAgeAwareEncodingState(history);
38
- ageAwareStateByHistory.set(history, created);
39
- return created;
40
- }
41
27
  /** 解析工具 arguments JSON;非法或空返 null(调用方据此降级到普通 preview)。 */
42
28
  function parseArgs(raw) {
43
29
  try {
@@ -47,49 +33,6 @@ function parseArgs(raw) {
47
33
  return null;
48
34
  }
49
35
  }
50
- /**
51
- * Thrashing 检测:同一工具 + 完全相同 arguments 在本轮重复 ≥ THRASH_THRESHOLD 次,
52
- * 返一段提示(注入到工具结果尾部),引导模型换思路而不是再试一次。
53
- * 阈值 2 = "试过两次同样的调用还没好,该停了"。指纹 = `${name}\\x00${args}`
54
- * (直接拼,不哈希——避免热路径开销;args 长度本身有限,内存压力可忽略)。
55
- * null 表示未触发,不污染输出。
56
- */
57
- const THRASH_THRESHOLD = 2;
58
- function thrashHint(name, args, count) {
59
- if (count < THRASH_THRESHOLD)
60
- return null;
61
- return (`\n\n[hint] This is call #${count} of \`${name}\` with identical arguments — ` +
62
- 'either failing or returning the same content. STOP retrying and switch strategy:\n' +
63
- '- read_file / glob → path likely wrong; call `glob` to discover paths, or `ask_human`\n' +
64
- '- run_command → Windows path-escaping issue; use `read_file` / `glob` with absolute paths instead\n' +
65
- '- write_file / edit_file CHANGE_CONFLICT → do not resend; read_file the same path and use its latest hash (use null only when read_file says the path is missing)\n' +
66
- '- edit_file old_string mismatch → re-read the exact region and copy it verbatim\n' +
67
- '- otherwise → re-read the tool description; the argument shape may be wrong');
68
- }
69
- /**
70
- * Detect only a strict streak of identical failures. Successful calls, or a
71
- * different tool/argument pair, end the streak. This keeps intentional phased
72
- * read_file/glob calls from being mislabeled after the underlying files change.
73
- */
74
- export function createThrashTracker() {
75
- let lastFailedFingerprint = null;
76
- let consecutiveFailures = 0;
77
- return (name, args, succeeded) => {
78
- if (succeeded) {
79
- lastFailedFingerprint = null;
80
- consecutiveFailures = 0;
81
- return null;
82
- }
83
- const fingerprint = `${name}\x00${args}`;
84
- if (fingerprint === lastFailedFingerprint)
85
- consecutiveFailures += 1;
86
- else {
87
- lastFailedFingerprint = fingerprint;
88
- consecutiveFailures = 1;
89
- }
90
- return thrashHint(name, args, consecutiveFailures);
91
- };
92
- }
93
36
  /** 只有显式声明 parallel 且无需权限确认的工具才进入普通并发组。 */
94
37
  function isParallelTool(name) {
95
38
  const tool = tools.find((candidate) => candidate.name === name);
@@ -158,58 +101,18 @@ function readDiffContext(tc, parsed) {
158
101
  }
159
102
  return { preWriteOld: null, editStartLine: 1 };
160
103
  }
161
- /** 回灌 tool 结果到 history:经 Context Optimization Pipeline 编码(tree/search/log/...)后裁到单条上限。
162
- * tool_call_id 与 assistant.tool_calls 按序配对。未注册 encoder 时回落 capToolResultForHistory(零行为变化)。
163
- * TUI 渲染(hooks.onToolResult)用原始 output,与此解耦——屏上看全量,LLM 看编码后紧凑版。
164
- * 出口再经 Relevance Pruner 做跨条裁剪:同 path 旧 read_file 自动 stub 为存根。
165
- * - pruner 在每个 runAgentCore 实例化一次(本闭包持有),会话级状态。
166
- * - 开关关闭时 pruner 不创建(零开销、零行为变化)。
167
- * 出口再经 Lifecycle Engine 做引用追踪:LIVE→REFERENCED→OBSOLETE→STUB 四态。
168
- * - lifecycle 也在每个 runAgentCore 实例化一次,登记 grep/glob/web_search/web_fetch 等 producer
169
- * 与 read/edit/write 的 consumer 关系;孤立+老化自动 STUB(观察类工具永不到 STUB)。
170
- * - 开关关闭时 lifecycle=null 完全跳过。 */
171
- /** RETRY-01: 把 thrash hint + 反思指针一次性拼到 output 尾部。
172
- * - thrash hint 永远可能存在(连续失败才出现);
173
- * - 反思指针仅在 error/denied 状态追加,success 跳过避免噪声。
174
- * 抽出 helper 是为了 sequential + parallel 两处共用,避免拼装逻辑漂移。
175
- * 返回追加后的字符串 + 反思 category(用于 QUAL-01 trace 硬事件),
176
- * success / aborted 时 category === null,调用方据此决定是否 emit 'retry_reflection' 事件。 */
177
- function appendRetryAnnotations(output, status, code, thrashHint) {
178
- let result = thrashHint ? `${output}${thrashHint}` : output;
179
- let reflectionCategory = null;
180
- if (status !== 'success' && status !== 'aborted') {
181
- reflectionCategory = classifyError(code);
182
- const reflection = reflectionHint(reflectionCategory);
183
- result = `${result}\n\n[retry reflection: ${reflectionCategory}]\n${reflection}`;
184
- }
185
- return { output: result, reflectionCategory };
186
- }
187
- /** ASK-01: ask_human 工具调用本 turn 计数器。纯函数无副作用,
188
- * 计数规则简单:1-based(本 turn 第 1 次 = 1, 第 2 次 = 2, 第 3+ 次 = exceeded)。
189
- * 预算上限常量与工作纪律段 "Budget: at most 2 ask_human calls per turn" 保持一致;
190
- * 改这里时,fixture 与纪律段都要同步。 */
191
- export const ASK_HUMAN_PER_TURN_BUDGET = 2;
192
- export function askHumanBudgetAnnotation(askHumanCountThisTurn, status) {
193
- // 工具失败 / aborted 不追加(避免噪声;模型已被错误消息告知失败)。
194
- if (status !== 'success')
195
- return null;
196
- if (askHumanCountThisTurn < ASK_HUMAN_PER_TURN_BUDGET)
197
- return null;
198
- if (askHumanCountThisTurn === ASK_HUMAN_PER_TURN_BUDGET) {
199
- return `\n\n[ask budget] This was your ${askHumanCountThisTurn}nd ask_human call this turn (budget = ${ASK_HUMAN_PER_TURN_BUDGET}). For the rest of this turn, prefer the safer default and disclose the choice in your final reply — do not silently guess.`;
200
- }
201
- // 第 3+ 次(超过预算):强硬提示,鼓励模型停下问自己是否还有意义。
202
- return `\n\n[ask budget EXCEEDED] This is ask_human call #${askHumanCountThisTurn} this turn (budget = ${ASK_HUMAN_PER_TURN_BUDGET}). Stop asking; choose a default, implement, and disclose the choice in your final reply. Continuing to ask is more harmful than a documented guess.`;
203
- }
104
+ /** 回灌 tool 结果到 history。
105
+ * 正常路径只做单条 hard cap;原始 output 同时供 TUI 展示,因此用户与模型
106
+ * 看到同一事实。Artifact/Relevance/Lifecycle 仅登记 metadata/provenance,
107
+ * 不在这里改写旧正文;所有自动清理与压缩统一由 80% pressure scheduler 决定。 */
204
108
  function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runtimeContextState = contextState, succeededOverride) {
205
109
  const succeeded = succeededOverride ?? isToolResultSuccess(output);
206
- const ageAware = config.contextOptimize ? ageAwareStateFor(history) : null;
207
- const encodingContext = ageAware?.preparePush(tc, succeeded);
208
110
  const msg = {
209
111
  role: 'tool',
210
112
  tool_call_id: tc.id,
211
- // 初次 push 始终保守(age=0);旧 Cold 结果在下一 step 的 sweep 中按类型降级。
212
- content: optimizeToolResult(tc.name, output, tc.arguments, encodingContext),
113
+ // Preserve evidence verbatim in normal operation; the hard per-result cap
114
+ // remains solely as a request-size safety rail.
115
+ content: capToolResultForHistory(tc.name, output),
213
116
  };
214
117
  history.push(msg);
215
118
  const messageIndex = history.length - 1;
@@ -225,8 +128,8 @@ function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runt
225
128
  * agent 核心循环(纯逻辑):
226
129
  * 流式调 LLM(经 hooks.onText 实时渲染)→ 有 tool_calls 就分组执行并回灌
227
130
  * → 否则流式正文即最终回复。history 在调用间持久,由调用方持有。
228
- * 步前经 session/maybeCompact 自动压缩(接近窗口上限时三层压缩);
229
- * 工具结果进 history 前经 Context Optimization Pipeline(optimizeToolResult:类型化编码 + 长度裁剪)。
131
+ * 步前由 session scheduler 检查真实 context pressure;达到 80% 时统一清理并压缩历史。
132
+ * 工具结果正常只经 capToolResultForHistory 的单条 hard safety cap。
230
133
  *
231
134
  * 中断语义:signal 经 executeTool(name, args, signal) 串进工具;run_command/web_fetch 等 abort 即时杀
232
135
  * (树杀子进程 / 取消 fetch),循环顶 if(signal.aborted) 兜底还原。不会留下未配对的 tool_call_id。
@@ -239,16 +142,8 @@ function pushToolResult(history, tc, output, pruner, lifecycle, _scheduler, runt
239
142
  export async function runAgentCore(opts) {
240
143
  const { history, userInput, signal, onContextUpdate, hooks } = opts;
241
144
  const runtimeContextState = opts.contextState ?? contextState;
242
- // ASK-01: 本 turn 已使用的 ask_human 次数(每次 tool call 后递增)。
243
- // 预算超过时,工具结果尾部追加 ask budget 提示,不直接拒绝调用(更轻量,
244
- // 也避免和现有 permission 系统的"拒绝"语义重叠)。
145
+ /** 本轮 ask_human 成功调用次数,仅用于 trace 观测,不影响工具执行或模型上下文。 */
245
146
  let askHumanCountThisTurn = 0;
246
- // PROMPT-02: 解析 preCompletionChecklist 选项。undefined = 启用默认 middleware;
247
- // false = opt-out(checklist 自身调试用);function = 调用方自定义。
248
- const _checklistMiddleware = createPreCompletionChecklistMiddleware();
249
- const preCompletionChecklistHandler = opts.preCompletionChecklist === false
250
- ? null
251
- : (opts.preCompletionChecklist ?? _checklistMiddleware.handler);
252
147
  const maxSteps = opts.maxSteps ?? config.maxSteps;
253
148
  // 中断还原:repl 的 /plan / /auto / Shift+Tab 等用户面触发 setAgentMode 中途切了模式,
254
149
  // abort 时连同模式一起还原回轮首。模型不再持有 switch_mode 工具,无法自切。
@@ -281,14 +176,13 @@ export async function runAgentCore(opts) {
281
176
  let done = false; // 正常完毕 / 达上限 true;中断 false(不显摘要)
282
177
  let traceStatus = 'error';
283
178
  let toolCallCount = 0;
284
- let latestValidation;
285
- const initialMutationVersion = getCurrentTurnMutationState().version;
286
- let validatedMutationVersion = initialMutationVersion;
287
- let fullyValidatedMutationVersion = initialMutationVersion;
288
- const validator = opts.validator ?? createAutomaticValidator();
289
179
  // 本轮 token 累计:每步 chat() 返回后把 result.usage 累加,供 onDone 摘要行 + AgentRunResult.usage
290
180
  // 透传给 repl(显示在底栏模式 chip 右边)。未开启 include_usage 或全失败时为 undefined。
291
181
  let turnUsage;
182
+ // 实时 chip ↻ 估算用:上一步 chat 实测 prompt(前缀缓存下当前步命中 ≈ 它)+ 后端是否报过 cache 命中
183
+ // (从不报 cache 的后端不估算,避免虚显 ↻)。
184
+ let lastStepPromptTokens = 0;
185
+ let providerCacheSeen = false;
292
186
  const addUsage = (u) => {
293
187
  if (!u)
294
188
  return;
@@ -303,48 +197,30 @@ export async function runAgentCore(opts) {
303
197
  : u;
304
198
  };
305
199
  const addToolUsage = (outcome) => addUsage(outcome.usage);
306
- // Only consecutive identical failures are thrashing. Any success (including
307
- // a mutation between two reads) or different call resets the streak.
308
- const recordAndHint = createThrashTracker();
309
200
  history.push({ role: 'user', content: userInput });
310
201
  // 中断回滚快照:push 用户消息后整段浅拷贝。abort 时 length=0;push(...saved) 还原。
311
202
  // 这样中断时至少保留用户消息(及之前的历史);每步工具全部执行完毕后刷新快照,
312
203
  // 保留已完成的 assistant+tool_calls+tool 结果,只丢弃当前未完成步骤的消息。
313
204
  // 用 slice() 而非 length:maybeCompact 会原地重建(length=0;push(...rebuilt)),savedLen 会失效。
314
205
  let savedHistory = history.slice();
315
- // drop_context 工具的上下文剔除回调:闭包捕获 history,原地剔除无关旧 tool 结果。
316
- // 保护由 dropContextFromHistory 内部保证:history[0](system)+ 当前轮(最后 user 及其后)永不剔除。
317
- // 子 agent 也在自己的 history 上操作(子 agent 独立 history),文件回滚事务则与主轮共享。
318
- const dropContext = (filter) => dropContextFromHistory(history, filter);
319
- // 相关性裁剪 pruner:每个 runAgentCore 实例一个,纯静态、不调 LLM、自动判定 read_file 失效。
320
- // 开关关闭时为 null,所有 pushToolResult 调用走无 pruner 路径(零行为变化)。
206
+ // Relevance and lifecycle collect provenance during normal work. Neither path
207
+ // rewrites history; exact supersession is applied only by the pressure scheduler.
321
208
  const relprune = config.contextRelprune ? createRelevancePruner() : null;
322
- // 观察者生命周期引擎:每个 runAgentCore 实例一个,纯静态、自动维护 grep/glob/web_search/web_fetch 等
323
- // producer 与 read/edit/write 的 consumer 引用关系;孤立+老化的非观察类工具自动 STUB。
324
- // 开关关闭时为 null,所有 pushToolResult / mutation 调用走无 lifecycle 路径(零行为变化)。
325
- // 引擎需要从已有会话 history 恢复观察结果的年龄和 path 索引;不能只追踪本次
326
- // runAgentCore,否则跨用户轮次的 grep/glob 永远不会衰减。
327
209
  let lifecycle = config.contextLifecycle
328
210
  ? createLifecycleEngine(history)
329
211
  : null;
330
212
  runtimeContextState.lifecycleStats = lifecycle?.stats();
331
213
  rehydrateArtifacts(runtimeContextState, history);
332
- // 预算调度器:每个 runAgentCore 实例一个,在 age-aware sweep 后评估并执行 warn / compact。
333
- // contextBudget 开关关闭时为 null。
214
+ // The scheduler is the sole automatic history-rewrite entry point. It runs
215
+ // superseded → stale artifact → old logs/search → compact at real pressure.
216
+ // contextBudget=false keeps only the infrastructure compact fallback.
334
217
  const scheduler = config.contextBudget !== false
335
- ? createBudgetScheduler(runtimeContextState) // 在 step 循环之外实例化一次,跨步持有 lastRunLog
218
+ ? createBudgetScheduler(runtimeContextState)
336
219
  : null;
337
- // age-aware encoder 与 history 数组同寿命;每轮重建一次以覆盖外部 /compact 等原地修改。
338
- const ageAware = config.contextOptimize ? ageAwareStateFor(history) : null;
339
- ageAware?.rehydrate(history);
340
220
  // 本轮流式状态:首个正文 token 到达即停 spinner(思考期间 spinner 持续转「思考中…」,不写思考内容)。
341
221
  let mode = 'idle';
342
222
  let gotText = false;
343
223
  let lastChar = '';
344
- // 早退重探:本 turn 已执行过工具但模型突然返回无工具调用 + 空文本 → 推一条提示让模型继续。
345
- // 短回复也可能是合法完成结果,不再按字符数误判;每 turn 最多触发 1 次以防死循环。
346
- let nudgeCount = 0;
347
- let hadToolsThisTurn = false;
348
224
  const onText = (s) => {
349
225
  hooks.onText?.(s); // 主 agent:走 markdown 渲染写内容区
350
226
  mode = 'text';
@@ -371,7 +247,6 @@ export async function runAgentCore(opts) {
371
247
  hooks.onAbort?.();
372
248
  history.length = 0;
373
249
  history.push(...savedHistory);
374
- ageAware?.rehydrate(history);
375
250
  setAgentMode(savedMode);
376
251
  };
377
252
  try {
@@ -390,7 +265,6 @@ export async function runAgentCore(opts) {
390
265
  terminationReason: 'aborted',
391
266
  finalText: null,
392
267
  usage: turnUsage,
393
- validation: latestValidation,
394
268
  changedFiles: mutation.changedFiles.map((item) => item.path),
395
269
  };
396
270
  }
@@ -402,10 +276,9 @@ export async function runAgentCore(opts) {
402
276
  const storedCalibration = getTokenCalibration(requestBaseURL, requestModel, activeTools);
403
277
  runtimeContextState.correction = storedCalibration.correction;
404
278
  runtimeContextState.calibrationSamples = storedCalibration.samples;
405
- // 初次 tool push 只做保守编码;预算评估前先对 Cold 且 age 达阈值的旧结果降级,
406
- // 避免 scheduler 根据马上会被 sweep 的陈旧占用误触发 history compact。
407
- ageAware?.sweep(history, userTurnBoundary(history, HOT_TURN_WINDOW));
408
- // 步前:五区 Budget Scheduler 在优化后的 history 上决策;开关关闭时退化回原 maybeCompact 路径。
279
+ // The scheduler is the only automatic path that may compress old evidence.
280
+ // Normal tool pushes and lifecycle tracking remain metadata-only.
281
+ // 步前:五区 Budget Scheduler 在当前完整 history 上决策;开关关闭时退化回 maybeCompact 路径。
409
282
  // 此时 spinner 已停,通知行干净。
410
283
  let historyRebuilt = false;
411
284
  const compactStartedAt = Date.now();
@@ -441,7 +314,6 @@ export async function runAgentCore(opts) {
441
314
  lifecycle = createLifecycleEngine(history);
442
315
  runtimeContextState.lifecycleStats = lifecycle.stats();
443
316
  }
444
- ageAware?.rehydrate(history);
445
317
  rehydrateArtifacts(runtimeContextState, history);
446
318
  }
447
319
  hooks.onStepStart?.(); // 主 agent:spinner.start('思考中')
@@ -452,12 +324,9 @@ export async function runAgentCore(opts) {
452
324
  const modelStartedAt = Date.now();
453
325
  const provider = safeProviderId(requestBaseURL);
454
326
  emitTrace('model_start', { model: requestModel, provider });
455
- const dynamicSystemSuffix = [
456
- opts.dynamicSystemSuffix?.().trim() ?? '',
457
- historyRebuilt
458
- ? '## Post-compaction recovery\nContext was compacted before this request. Re-establish the current objective and unresolved work from retained evidence or the session note, avoid repeating completed investigation, and re-read exact file context before any dependent edit.'
459
- : '',
460
- ].filter(Boolean).join('\n\n');
327
+ const dynamicSystemSuffix = historyRebuilt
328
+ ? '## Post-compaction recovery\nContext was compacted before this request. Re-establish the current objective and unresolved work from retained evidence or the session note, avoid repeating completed investigation, and re-read exact file context before any dependent edit.'
329
+ : '';
461
330
  const systemMessage = history[0];
462
331
  const requestHistory = dynamicSystemSuffix
463
332
  && systemMessage?.role === 'system'
@@ -470,10 +339,32 @@ export async function runAgentCore(opts) {
470
339
  ...history.slice(1),
471
340
  ]
472
341
  : history;
342
+ // 实时用量:当前步 prompt 估算(含校准系数)+ 流式累计 completion 估算,
343
+ // 叠上已完成步的实测 turnUsage,经 onLiveUsage 推给底栏实时 chip。
344
+ // turnUsage 在闭包里被 addUsage 原地更新,reportLive 每次调用读最新值。
345
+ const stepPromptEst = estimatePromptTokens(requestHistory, activeTools, runtimeContextState.correction);
346
+ const reportLive = (p) => {
347
+ // 当前步 prompt:末尾 usage chunk 到达后用实测,流式期间用估算(含校准)。
348
+ // 当前步 cache 命中同理:上报即用实测;流式期间按前缀缓存估算 ≈ 上一步实测 prompt
349
+ // (当前 prompt 总含其为前缀),不超过当前步 prompt;后端从不报 cache 时不估算。
350
+ // 口径与轮末摘要一致:chip ↑ 显计费 prompt(裸 - cached),↓/↻ 同。
351
+ const curPrompt = p.promptTokens ?? stepPromptEst;
352
+ const curCached = p.cachedTokens ?? (providerCacheSeen
353
+ ? Math.min(lastStepPromptTokens, curPrompt)
354
+ : 0);
355
+ hooks.onLiveUsage?.({
356
+ promptTokens: (turnUsage?.promptTokens ?? 0) + curPrompt,
357
+ completionTokens: (turnUsage?.completionTokens ?? 0) + p.completionTokens,
358
+ totalTokens: (turnUsage?.totalTokens ?? 0) + curPrompt + p.completionTokens,
359
+ cachedTokens: (turnUsage?.cachedTokens ?? 0) + curCached,
360
+ });
361
+ };
362
+ reportLive({ completionTokens: 0 }); // 思考阶段先显 ↑ prompt 估算,首 token 到达后 ↓ 开始涨
473
363
  try {
474
364
  result = await chat(requestHistory, {
475
365
  onText,
476
366
  onToolCall,
367
+ onProgress: reportLive,
477
368
  onRetry: (retry) => emitTrace('model_retry', {
478
369
  model: requestModel,
479
370
  provider,
@@ -510,7 +401,6 @@ export async function runAgentCore(opts) {
510
401
  terminationReason: 'aborted',
511
402
  finalText: null,
512
403
  usage: turnUsage,
513
- validation: latestValidation,
514
404
  changedFiles: mutation.changedFiles.map((item) => item.path),
515
405
  };
516
406
  }
@@ -529,6 +419,11 @@ export async function runAgentCore(opts) {
529
419
  });
530
420
  runtimeContextState.lastUsage = result.usage; // 供 /context 与状态行显示实测 token
531
421
  addUsage(result.usage); // 本轮累计:onDone 摘要行 + AgentRunResult.usage 透传
422
+ if (result.usage) {
423
+ lastStepPromptTokens = result.usage.promptTokens; // 下一步流式期 ↻ 估算的前缀基准
424
+ if (result.usage.cachedTokens > 0)
425
+ providerCacheSeen = true;
426
+ }
532
427
  // 用本次实际发送的 tools 计算分母,再以 EWMA 更新 provider/model/tool-set 校准。
533
428
  // 只持久化比例与样本数;无 usage 或短 prompt 时保持既有值。
534
429
  if (result.usage?.promptTokens && result.usage.promptTokens > 100) {
@@ -542,7 +437,6 @@ export async function runAgentCore(opts) {
542
437
  onContextUpdate?.();
543
438
  if (result.toolCalls.length > 0) {
544
439
  toolCallCount += result.toolCalls.length;
545
- hadToolsThisTurn = true;
546
440
  // 流式正文末尾补换行(若 onToolCall 已补则 lastChar='\n',此处 no-op);防 ● 行黏在正文行尾
547
441
  if (mode !== 'idle' && lastChar !== '\n')
548
442
  hooks.onTextEnd?.();
@@ -556,12 +450,23 @@ export async function runAgentCore(opts) {
556
450
  function: { name: tc.name, arguments: tc.arguments },
557
451
  })),
558
452
  });
453
+ // Record interstitial narration for observability only. It never changes tool output
454
+ // or injects instructions back into the model context.
455
+ const narration = result.content?.trim() ?? '';
456
+ if (narration) {
457
+ emitTrace('narration', {
458
+ chars: [...narration].length,
459
+ toolCalls: result.toolCalls.length,
460
+ step,
461
+ });
462
+ }
559
463
  // 工具分组执行(保 tool_calls 原顺序):safe parallel 工具照常并发;连续
560
464
  // resource-locked mutation 先按序完成权限预检,再按 canonical resource lock 启动。
561
465
  // registry 对所有真实资源访问统一持锁,所以不同 Agent 间的 read/write/process 也不会竞态。
562
466
  // 串行工具仍是本调用列表内的屏障;渲染/history 回灌始终按原 tool_calls 顺序。
563
467
  // executeToolOutcome 永不抛错,失败通过结构化 status/code 返回。
564
468
  const calls = result.toolCalls;
469
+ const modelAttachments = [];
565
470
  const tracedCalls = calls.map((tc, index) => ({
566
471
  toolCallId: `${traceTurnId}:step:${step}:tool:${index}`,
567
472
  args: summarizeToolArguments(tc.arguments),
@@ -573,28 +478,15 @@ export async function runAgentCore(opts) {
573
478
  tool: tc.name,
574
479
  argumentHash: traceCall.args.sha256,
575
480
  arguments: traceCall.args,
576
- attempt: 1,
577
- retry: 0,
578
481
  }, {
579
482
  toolCallId: traceCall.toolCallId,
580
483
  ...(tc.id ? { providerToolCallId: tc.id } : {}),
581
484
  });
582
485
  }
583
- const traceToolRetry = (tc, index, retry) => {
584
- const traceCall = tracedCalls[index];
585
- emitTrace('tool_retry', {
586
- tool: tc.name,
587
- argumentHash: traceCall.args.sha256,
588
- attempt: retry.attempt,
589
- nextAttempt: retry.nextAttempt,
590
- waitMs: retry.waitMs,
591
- code: retry.code,
592
- }, {
593
- toolCallId: traceCall.toolCallId,
594
- ...(tc.id ? { providerToolCallId: tc.id } : {}),
595
- });
596
- };
597
486
  const traceToolEnd = (tc, index, outcome) => {
487
+ if (outcome.status === 'success' && outcome.modelAttachments?.length) {
488
+ modelAttachments.push(...outcome.modelAttachments);
489
+ }
598
490
  const traceCall = tracedCalls[index];
599
491
  emitTrace('tool_call_end', {
600
492
  tool: tc.name,
@@ -603,9 +495,6 @@ export async function runAgentCore(opts) {
603
495
  code: outcome.code,
604
496
  retryable: outcome.retryable,
605
497
  durationMs: outcome.durationMs ?? 0,
606
- attempt: outcome.attempts ?? 1,
607
- retry: Math.max(0, (outcome.attempts ?? 1) - 1),
608
- retryDelayMs: outcome.retryDelayMs ?? 0,
609
498
  changedFiles: outcome.changedFiles ?? [],
610
499
  staleFiles: outcome.staleFiles ?? [],
611
500
  ...(outcome.changeSet ? { changeSet: outcome.changeSet } : {}),
@@ -630,8 +519,7 @@ export async function runAgentCore(opts) {
630
519
  durationMs: 0,
631
520
  };
632
521
  hooks.onToolResult?.(currentCall, error, null, null, 1);
633
- const hint = recordAndHint(currentCall.name, currentCall.arguments, false);
634
- pushToolResult(history, currentCall, hint ? `${error}${hint}` : error, relprune, lifecycle, scheduler, runtimeContextState, false);
522
+ pushToolResult(history, currentCall, error, relprune, lifecycle, scheduler, runtimeContextState, false);
635
523
  traceToolEnd(currentCall, i, outcome);
636
524
  i++;
637
525
  continue;
@@ -649,10 +537,7 @@ export async function runAgentCore(opts) {
649
537
  for (const tc of batch)
650
538
  hooks.onToolHeader?.(tc);
651
539
  hooks.onToolStart?.(batch[0].name);
652
- const started = batch.map((tc, offset) => executeToolOutcome(tc.name, tc.arguments, signal, {
653
- dropContext,
654
- onRetry: (retry) => traceToolRetry(tc, i + offset, retry),
655
- }));
540
+ const started = batch.map((tc) => executeToolOutcome(tc.name, tc.arguments, signal));
656
541
  for (let k = 0; k < batch.length; k++) {
657
542
  const tc = batch[k];
658
543
  const outcome = await started[k];
@@ -661,32 +546,15 @@ export async function runAgentCore(opts) {
661
546
  traceToolEnd(tc, i + k, outcome);
662
547
  const output = outcome.output;
663
548
  hooks.onToolResult?.(tc, output, null, null, 1); // 并行工具无 diff
664
- // Thrashing:history 里附 hint(UI 已用干净 output 渲染,避免屏幕噪声)
665
- const hint = recordAndHint(tc.name, tc.arguments, outcome.status === 'success');
666
- const { output: annotated, reflectionCategory } = appendRetryAnnotations(output, outcome.status, outcome.code, hint);
667
- // QUAL-01 trace 硬事件:反思指针注入计数(history 文本扫描会因 LLM 在
668
- // 正文里提一句 "[retry reflection]" 而误算,改用 core 接缝处显式 emit)。
669
- if (reflectionCategory !== null) {
670
- emitTrace('retry_reflection', {
671
- tool: tc.name,
672
- code: outcome.code,
673
- category: reflectionCategory,
674
- attempt: outcome.attempts ?? 1,
675
- }, tc.id ? { providerToolCallId: tc.id } : {});
676
- }
677
- // ASK-01: 计数 + 预算提示
678
549
  if (tc.name === 'ask_human' && outcome.status === 'success') {
679
550
  askHumanCountThisTurn += 1;
680
- // QUAL-01 trace 硬事件:ask_human 真实调用计数(取代 tool_call_end.name 推断)。
681
551
  emitTrace('ask_human_call', {
682
552
  tool: tc.name,
683
553
  status: outcome.status,
684
554
  perTurnCount: askHumanCountThisTurn,
685
555
  }, tc.id ? { providerToolCallId: tc.id } : {});
686
556
  }
687
- const askBudget = askHumanBudgetAnnotation(askHumanCountThisTurn, outcome.status);
688
- const finalOutput = askBudget ? `${annotated}${askBudget}` : annotated;
689
- pushToolResult(history, tc, finalOutput, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
557
+ pushToolResult(history, tc, output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
690
558
  }
691
559
  hooks.onToolDone?.();
692
560
  i = j;
@@ -739,14 +607,12 @@ export async function runAgentCore(opts) {
739
607
  const firstAllowed = entries.find((entry) => !entry.denied);
740
608
  if (firstAllowed)
741
609
  hooks.onToolStart?.(firstAllowed.tc.name);
742
- const started = entries.map((entry, offset) => entry.denied
610
+ const started = entries.map((entry) => entry.denied
743
611
  ? Promise.resolve(entry.denied)
744
612
  : executeToolOutcome(entry.tc.name, entry.tc.arguments, signal, {
745
- dropContext,
746
613
  onLockAcquired: (lockedArgs) => {
747
614
  entry.diff = readDiffContext(entry.tc, lockedArgs);
748
615
  },
749
- onRetry: (retry) => traceToolRetry(entry.tc, i + offset, retry),
750
616
  }));
751
617
  for (let k = 0; k < entries.length; k++) {
752
618
  const entry = entries[k];
@@ -755,8 +621,7 @@ export async function runAgentCore(opts) {
755
621
  opts.onToolOutcome?.(entry.tc.name, entry.parsed ?? {}, outcome);
756
622
  traceToolEnd(entry.tc, i + k, outcome);
757
623
  hooks.onToolResult?.(entry.tc, outcome.output, entry.denied ? null : entry.parsed, entry.diff.preWriteOld, entry.diff.editStartLine);
758
- const hint = recordAndHint(entry.tc.name, entry.tc.arguments, outcome.status === 'success');
759
- pushToolResult(history, entry.tc, hint ? `${outcome.output}${hint}` : outcome.output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
624
+ pushToolResult(history, entry.tc, outcome.output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
760
625
  const invalidatedFiles = [...new Set([
761
626
  ...(outcome.changedFiles ?? []),
762
627
  ...(outcome.staleFiles ?? []),
@@ -791,9 +656,7 @@ export async function runAgentCore(opts) {
791
656
  durationMs: 0,
792
657
  };
793
658
  hooks.onToolResult?.(tc, err, null, null, 1);
794
- // Thrashing:同上
795
- const hint = recordAndHint(tc.name, tc.arguments, false);
796
- pushToolResult(history, tc, hint ? `${err}${hint}` : err, relprune, lifecycle, scheduler);
659
+ pushToolResult(history, tc, err, relprune, lifecycle, scheduler);
797
660
  traceToolEnd(tc, i, outcome);
798
661
  i++;
799
662
  continue;
@@ -822,8 +685,7 @@ export async function runAgentCore(opts) {
822
685
  hooks.onToolHeader?.(tc);
823
686
  const outcome = deniedOutcome(tc.name);
824
687
  hooks.onToolResult?.(tc, outcome.output, null, null, 1);
825
- const hint = recordAndHint(tc.name, tc.arguments, false);
826
- pushToolResult(history, tc, hint ? `${outcome.output}${hint}` : outcome.output, relprune, lifecycle, scheduler, runtimeContextState, false);
688
+ pushToolResult(history, tc, outcome.output, relprune, lifecycle, scheduler, runtimeContextState, false);
827
689
  traceToolEnd(tc, i, outcome);
828
690
  i++;
829
691
  continue;
@@ -836,12 +698,10 @@ export async function runAgentCore(opts) {
836
698
  let diff = readDiffContext(tc, mutationParsed);
837
699
  hooks.onToolStart?.(tc.name);
838
700
  const outcome = await executeToolOutcome(tc.name, tc.arguments, signal, {
839
- dropContext,
840
701
  onLockAcquired: (lockedArgs) => {
841
702
  if (mutationParsed)
842
703
  diff = readDiffContext(tc, lockedArgs);
843
704
  },
844
- onRetry: (retry) => traceToolRetry(tc, i, retry),
845
705
  });
846
706
  addToolUsage(outcome);
847
707
  opts.onToolOutcome?.(tc.name, parsed ?? {}, outcome);
@@ -849,33 +709,15 @@ export async function runAgentCore(opts) {
849
709
  const output = outcome.output;
850
710
  hooks.onToolDone?.();
851
711
  hooks.onToolResult?.(tc, output, mutationParsed, diff.preWriteOld, diff.editStartLine);
852
- // Thrashing:同上(history 附 hint,UI 干净)
853
- const hint = recordAndHint(tc.name, tc.arguments, outcome.status === 'success');
854
- // RETRY-01: 同 parallel 路径,在非成功状态追加反思指针。
855
- const { output: annotated, reflectionCategory } = appendRetryAnnotations(output, outcome.status, outcome.code, hint);
856
- // QUAL-01 trace 硬事件:反思指针注入计数(history 文本扫描会因 LLM 在
857
- // 正文里提一句 "[retry reflection]" 而误算,改用 core 接缝处显式 emit)。
858
- if (reflectionCategory !== null) {
859
- emitTrace('retry_reflection', {
860
- tool: tc.name,
861
- code: outcome.code,
862
- category: reflectionCategory,
863
- attempt: outcome.attempts ?? 1,
864
- }, tc.id ? { providerToolCallId: tc.id } : {});
865
- }
866
- // ASK-01: ask_human 计数 + 预算提示(同 parallel 路径)。
867
712
  if (tc.name === 'ask_human' && outcome.status === 'success') {
868
713
  askHumanCountThisTurn += 1;
869
- // QUAL-01 trace 硬事件:ask_human 真实调用计数(取代 tool_call_end.name 推断)。
870
714
  emitTrace('ask_human_call', {
871
715
  tool: tc.name,
872
716
  status: outcome.status,
873
717
  perTurnCount: askHumanCountThisTurn,
874
718
  }, tc.id ? { providerToolCallId: tc.id } : {});
875
719
  }
876
- const askBudget = askHumanBudgetAnnotation(askHumanCountThisTurn, outcome.status);
877
- const finalOutput = askBudget ? `${annotated}${askBudget}` : annotated;
878
- pushToolResult(history, tc, finalOutput, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
720
+ pushToolResult(history, tc, output, relprune, lifecycle, scheduler, runtimeContextState, outcome.status === 'success');
879
721
  const invalidatedFiles = [...new Set([
880
722
  ...(outcome.changedFiles ?? []),
881
723
  ...(outcome.staleFiles ?? []),
@@ -891,6 +733,29 @@ export async function runAgentCore(opts) {
891
733
  i++;
892
734
  }
893
735
  }
736
+ if (modelAttachments.length > 0) {
737
+ const names = modelAttachments.map((attachment) => attachment.name).join(', ');
738
+ const content = [
739
+ {
740
+ type: 'text',
741
+ text: `The view_image tool loaded the following visual input: ${names}. Analyze the attached image content directly.`,
742
+ },
743
+ ...modelAttachments.map((attachment) => ({
744
+ type: 'image_url',
745
+ image_url: {
746
+ url: attachment.dataUrl,
747
+ // 不默认补 auto: OpenAI 省略时等同 auto,但 MiniMax 仅接受 low/default/high。
748
+ // 让各 provider 采用默认枚举;仅保留工具明确请求的 low/high。
749
+ ...(attachment.detail === 'low' || attachment.detail === 'high'
750
+ ? { detail: attachment.detail }
751
+ : {}),
752
+ },
753
+ })),
754
+ ];
755
+ // OpenAI tool-call protocol requires every tool result to immediately follow the
756
+ // assistant tool_calls message; append visual input only after the full batch.
757
+ history.push({ role: 'user', content });
758
+ }
894
759
  // 工具步末尾补一空行:与下一轮的思考 / 正文分隔(否则 ↳ 后紧接 ▎ 思考,无空行不好看;
895
760
  // 与正文→● 的 1 空行对称)。工具结果已以 \n 收尾,此处再补 \n 恰好 1 空行。
896
761
  hooks.onToolBatchEnd?.();
@@ -901,169 +766,20 @@ export async function runAgentCore(opts) {
901
766
  }
902
767
  if (mode !== 'idle' && lastChar !== '\n')
903
768
  hooks.onTextEnd?.(); // 流式末尾补换行
904
- // 早退保护:本 turn 已执行过工具调用,但模型突然返回无工具 + 极短/空文本 → 很可能是在探索中途
905
- // 提前"说完了"。此时推一条 user 提示消息让模型继续探索,而非直接退出。每 turn 最多 1 次,防死循环。
906
- // 早退重探只处理真正的空回复。短回复可能是合法完成结果(例如精确状态标记);
907
- // 仅凭字符数继续调用会重复输出,并产生一次完整的额外模型请求。
908
- const replyIsEmpty = (result.content?.trim().length ?? 0) === 0;
909
- if (hadToolsThisTurn && nudgeCount < 1 && replyIsEmpty) {
910
- nudgeCount++;
911
- history.push({ role: 'assistant', content: result.content });
912
- history.push({
913
- role: 'user',
914
- content: 'You stopped before completing the task. Please continue investigating — call more tools if needed, or provide a complete answer based on what you have gathered so far.',
915
- });
916
- continue; // 带着提示再调一次 LLM
917
- }
918
- // 没有工具调用:候选正文已流式打印。若本轮有新的代码变更,先通过框架验证门;
919
- // failed 作为 system observation 风格的 user 消息回灌,不能伪造无配对的 tool 消息。
769
+ // 没有工具调用:接受 agent 的完成判断。框架不自动运行测试、构建或完成门,
770
+ // 也不因缺少验证证据强制追加模型轮次;agent 仍可自行调用工具验证。
920
771
  if (!gotText)
921
772
  hooks.onNoReply?.();
922
- const candidate = { role: 'assistant', content: result.content };
923
- const mutationBeforeValidation = getCurrentTurnMutationState();
924
- const shouldValidate = opts.autoValidate === true &&
925
- getAgentMode() !== 'plan' &&
926
- (mutationBeforeValidation.version > validatedMutationVersion || latestValidation?.status === 'failed');
927
- if (shouldValidate) {
928
- history.push(candidate);
929
- emitTrace('validation_start', {
930
- mutationVersion: mutationBeforeValidation.version,
931
- changedFiles: mutationBeforeValidation.changedFiles.map((item) => item.path),
932
- });
933
- try {
934
- latestValidation = await validator(signal, {
935
- onCommandStart: (command) => hooks.onValidationStart?.(command),
936
- onPermissionDecision: (permission) => {
937
- const summary = summarizeToolArguments(JSON.stringify(permission.arguments));
938
- emitTrace('permission', {
939
- source: 'automatic_validation',
940
- tool: permission.tool,
941
- decision: permission.decision,
942
- argumentHash: summary.sha256,
943
- });
944
- },
945
- });
946
- }
947
- catch (error) {
948
- const mutation = getCurrentTurnMutationState();
949
- const message = `Automatic validation failed to run: ${error instanceof Error ? error.message : String(error)}`;
950
- const status = signal?.aborted ? 'aborted' : 'failed';
951
- latestValidation = {
952
- status,
953
- level: 'V0',
954
- output: message,
955
- durationMs: 0,
956
- diagnostics: [{
957
- level: 'V0', source: 'verifier', severity: 'error', code: 'VERIFIER_ERROR', message,
958
- }],
959
- stages: [],
960
- verificationComplete: false,
961
- fingerprint: `verifier-error-${mutation.version}`,
962
- inputFingerprint: `mutation-${mutation.version}`,
963
- inputMutationVersion: mutation.version,
964
- affectedPackages: [],
965
- changedFiles: mutation.changedFiles.map((item) => item.path),
966
- mutationVersion: mutation.version,
967
- };
968
- }
969
- emitTrace('validation_end', {
970
- status: latestValidation.status,
971
- level: latestValidation.level,
972
- durationMs: latestValidation.durationMs,
973
- verificationComplete: latestValidation.verificationComplete,
974
- skipReason: latestValidation.skipReason,
975
- fingerprint: latestValidation.fingerprint,
976
- mutationVersion: latestValidation.mutationVersion,
977
- stages: latestValidation.stages.map((stage) => ({
978
- level: stage.level,
979
- status: stage.status,
980
- adapter: stage.adapter,
981
- code: stage.diagnostics[0]?.code,
982
- durationMs: stage.durationMs,
983
- cached: stage.cached === true,
984
- })),
985
- });
986
- hooks.onValidationResult?.(latestValidation);
987
- validatedMutationVersion = latestValidation.mutationVersion;
988
- if (latestValidation.status === 'passed' && latestValidation.verificationComplete) {
989
- fullyValidatedMutationVersion = latestValidation.mutationVersion;
990
- }
991
- if (signal?.aborted || latestValidation.status === 'aborted') {
992
- abortRestore();
993
- traceStatus = 'aborted';
994
- const mutation = getCurrentTurnMutationState();
995
- return {
996
- completed: false,
997
- terminationReason: 'aborted',
998
- finalText: null,
999
- usage: turnUsage,
1000
- validation: latestValidation,
1001
- changedFiles: mutation.changedFiles.map((item) => item.path),
1002
- };
1003
- }
1004
- if (latestValidation.status === 'failed') {
1005
- history.push({
1006
- role: 'user',
1007
- content: '[System observation: automatic validation failed]\n' +
1008
- `Command: ${latestValidation.command ?? '(internal verifier)'}\n` +
1009
- `${latestValidation.output}\n\n` +
1010
- 'Fix the reported problem, then finish the task. Do not claim success until validation passes.',
1011
- });
1012
- // 验证及其失败观察均已完整落入 history;下一步中断时可安全保留。
1013
- savedHistory = history.slice();
1014
- continue;
1015
- }
1016
- }
1017
- else {
1018
- // PROMPT-02: 硬关卡 — 收尾路径上,如果本 turn 有 mutation 且未通过
1019
- // validation(autoValidate=false / 验证已 passed 但 LLM 又宣称完成),
1020
- // 推一条 [checklist] user 消息,强制模型再走一轮显式确认。
1021
- // 防死循环:checklist 已推过 2 次仍无新工具调用 → 放行(走原本的 push)。
1022
- const finalMutationForChecklist = getCurrentTurnMutationState();
1023
- const checklistCtx = {
1024
- hadMutation: finalMutationForChecklist.version > 0,
1025
- lastValidationStatus: (latestValidation?.status === 'failed' || latestValidation?.status === 'passed')
1026
- ? latestValidation.status
1027
- : 'none',
1028
- mode: getAgentMode(),
1029
- modelFamily: inferModelFamily(config.model),
1030
- };
1031
- const checklistRetryCount = history.__checklistStreak ?? 0;
1032
- const shouldFireChecklist = preCompletionChecklistHandler !== null
1033
- && preCompletionChecklistHandler(checklistCtx)
1034
- && checklistRetryCount < 2;
1035
- if (shouldFireChecklist) {
1036
- history.push(candidate);
1037
- history.push({
1038
- role: 'user',
1039
- content: _checklistMiddleware.buildUserMessage(checklistCtx.modelFamily),
1040
- });
1041
- history.__checklistStreak = checklistRetryCount + 1;
1042
- // QUAL-01 trace 硬事件:PROMPT-02 触发计数(取代 history 文本扫描
1043
- // [checklist] marker — LLM 在正文里提一句 "[checklist]" 也会被误算)。
1044
- emitTrace('checklist_triggered', {
1045
- streak: checklistRetryCount + 1,
1046
- validationStatus: checklistCtx.lastValidationStatus,
1047
- hadMutation: checklistCtx.hadMutation,
1048
- modelFamily: checklistCtx.modelFamily ?? 'other',
1049
- });
1050
- // 重要: 不置 done,让主循环继续下一轮(模型被迫用工具或写出可验证声明)。
1051
- continue;
1052
- }
1053
- history.push(candidate);
1054
- }
773
+ history.push({ role: 'assistant', content: result.content });
1055
774
  const finalMutation = getCurrentTurnMutationState();
1056
775
  done = true;
1057
776
  traceStatus = 'completed';
1058
- const verified = finalMutation.version <= fullyValidatedMutationVersion;
1059
777
  return {
1060
778
  completed: true,
1061
- terminationReason: verified ? 'completed' : 'completed_unverified',
779
+ terminationReason: 'completed',
1062
780
  finalText: result.content,
1063
781
  usage: turnUsage,
1064
- validation: latestValidation,
1065
782
  changedFiles: finalMutation.changedFiles.map((item) => item.path),
1066
- history: history.slice(),
1067
783
  };
1068
784
  }
1069
785
  finally {
@@ -1077,15 +793,12 @@ export async function runAgentCore(opts) {
1077
793
  done = true;
1078
794
  traceStatus = 'max_steps';
1079
795
  const finalMutation = getCurrentTurnMutationState();
1080
- const hasUnverifiedChanges = finalMutation.version > fullyValidatedMutationVersion;
1081
796
  return {
1082
797
  completed: false,
1083
- terminationReason: hasUnverifiedChanges ? 'unverified_changes' : 'max_steps',
798
+ terminationReason: 'max_steps',
1084
799
  finalText: null,
1085
800
  usage: turnUsage,
1086
- validation: latestValidation,
1087
801
  changedFiles: finalMutation.changedFiles.map((item) => item.path),
1088
- history: history.slice(),
1089
802
  };
1090
803
  }
1091
804
  finally {
@@ -1097,7 +810,6 @@ export async function runAgentCore(opts) {
1097
810
  toolCalls: toolCallCount,
1098
811
  changedFiles: finalMutation.changedFiles.map((item) => item.path),
1099
812
  totalTokens: turnUsage?.totalTokens,
1100
- validationStatus: latestValidation?.status,
1101
813
  });
1102
814
  try {
1103
815
  opts.onTrace?.({
@@ -1109,7 +821,6 @@ export async function runAgentCore(opts) {
1109
821
  toolCalls: toolCallCount,
1110
822
  changedFiles: finalMutation.changedFiles.map((item) => item.path),
1111
823
  usage: turnUsage,
1112
- validation: latestValidation,
1113
824
  });
1114
825
  }
1115
826
  catch {