@miphamai/cli 0.81.9 → 0.83.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@miphamai/cli",
3
- "version": "0.81.9",
3
+ "version": "0.83.0",
4
4
  "description": "Mipham Code — Multi-model open-core intelligent coding terminal by MiphamAI",
5
5
  "keywords": [
6
6
  "ai",
@@ -13,7 +13,7 @@
13
13
  */
14
14
 
15
15
  import { randomUUID } from 'node:crypto'
16
- import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
16
+ import { CrsiSandbox, validateBlastRadius, validateMergeConvergence } from './crsi-sandbox'
17
17
  import type { CrsiModificationResult } from './crsi-sandbox'
18
18
  import { appendEvalScore, getLastEvalScore, regressedAnchors } from './eval-harness'
19
19
  import { mechanismSentinel, type RewardFn } from './reward-fn'
@@ -36,6 +36,19 @@ export interface CrsiProposal {
36
36
  * 局部正确、全局遗漏。自修改前必须摸清并声明全部受影响路径,否则 fail-closed 拒绝。
37
37
  */
38
38
  blastRadius?: string[]
39
+ /**
40
+ * ε:提交者**事前**写下的预期效果(任务表现提升点数)。
41
+ * 缺席 = 不预测。判定见 improvement-track 的 predictionHit。
42
+ */
43
+ expectedEffect?: number
44
+ /** R:风险声明(这次改动可能在哪方面变差)。缺席 = 未声明。 */
45
+ risk?: string
46
+ /**
47
+ * 声明这是一次**合并型**提案(整合已有内容,而非新增)。
48
+ * 只有它为 true 时 `validateMergeConvergence` 才开火 —— 学习本身就是增长,
49
+ * 对新增型设非增长约束等于永久禁掉 `/crsi propose`。
50
+ */
51
+ merge?: boolean
39
52
  }
40
53
 
41
54
  // ── Pending proposal registry (两阶段闸门) ──
@@ -70,6 +83,26 @@ export async function runCrsiModification(
70
83
  }
71
84
  }
72
85
 
86
+ // B_H 收敛闸:合并型提案不得抬高脚手架成本。
87
+ // 位置与 blast radius 闸同序 —— 都在 worktree 之前,纯字符串比较、零磁盘 I/O、零副作用。
88
+ const convergenceError = validateMergeConvergence(proposal)
89
+ if (convergenceError) {
90
+ return {
91
+ modification: {
92
+ id: 'crsi-mod-rejected-merge-convergence',
93
+ description: proposal.description,
94
+ filePath: proposal.filePath,
95
+ newContent: proposal.newContent,
96
+ originalContent: proposal.originalContent ?? '',
97
+ crsiInsightId: proposal.crsiInsightId,
98
+ timestamp: new Date().toISOString(),
99
+ },
100
+ applied: false,
101
+ phase: 'failed',
102
+ error: convergenceError,
103
+ }
104
+ }
105
+
73
106
  sandbox.createWorktree()
74
107
 
75
108
  const applied = sandbox.applyModification({
@@ -336,7 +336,7 @@ export async function selectTargetSkill(
336
336
  return extractFilePath(response, skillFiles)
337
337
  }
338
338
 
339
- const PROSE_GENERATE_PROMPT_VERSION = '1.0.0'
339
+ const PROSE_GENERATE_PROMPT_VERSION = '1.1.0'
340
340
 
341
341
  function buildGenerateProsePrompt(
342
342
  signal: CrsiSignal,
@@ -357,7 +357,12 @@ function buildGenerateProsePrompt(
357
357
  '当前内容:',
358
358
  originalContent,
359
359
  '',
360
- '请返回改进后的完整 markdown(保持 YAML frontmatter 的 name/description 字段,正文针对失败信号做针对性改进)。只返回 markdown,不要额外说明。',
360
+ '返回格式(严格遵守,两段):',
361
+ '第 1 行:一行 JSON,写下你对这次改动的**预期效果**与**风险**:',
362
+ '{"expectedDelta": <number 或 null>, "risk": "<字符串>"}',
363
+ '- expectedDelta 是预期该 skill 的任务表现提升**点数**(可正可负;无法预测写 null)。',
364
+ '- risk 是这次改动可能在哪方面变差(一句话)。',
365
+ '第 2 行起:改进后的完整 markdown(保持 YAML frontmatter 的 name/description 字段,正文针对失败信号做针对性改进)。不要用代码围栏包住。',
361
366
  ].join('\n')
362
367
  }
363
368
 
@@ -366,16 +371,68 @@ function stripMarkdownFence(text: string): string {
366
371
  return match ? match[1]! : text
367
372
  }
368
373
 
374
+ /** prose 提议的解析产物:正文 + 可选的事前预登记(ε 与风险声明)。 */
375
+ export interface ProsePrediction {
376
+ body: string
377
+ /** ε:事前写下的预期提升点数。缺席 = 模型没预测(含显式写 null)。 */
378
+ expectedEffect?: number
379
+ /** R:风险声明。缺席 = 未声明。 */
380
+ risk?: string
381
+ }
382
+
383
+ /**
384
+ * 解析 prose 响应:可选的一行 JSON 前缀(ε)+ 正文。
385
+ *
386
+ * 顺序是**先归一化、后嗅探**(不可颠倒):stripMarkdownFence 的正则锚在串首
387
+ * (/^```(?:markdown|md)?\s*\n…\n```\s*$/)。若先剥「首行围栏」再嗅探,正文尾部的
388
+ * 那个 ``` 就再没有东西去剥它 ⇒ 孤立的尾部围栏会进入写盘路径。
389
+ *
390
+ * 认领标记是**含 `expectedDelta` 键**(盖住 number 与显式 null 两种写法);
391
+ * 其余任何情况都走兜底 —— 正文 = 归一化后的原文,一字不改。
392
+ */
393
+ export function parseProsePrediction(raw: string): ProsePrediction {
394
+ const stripped = stripMarkdownFence(raw)
395
+ const lines = stripped.split('\n')
396
+ const firstIdx = lines.findIndex((l) => l.trim() !== '')
397
+ if (firstIdx === -1) return { body: stripped }
398
+
399
+ let parsed: unknown
400
+ try {
401
+ parsed = JSON.parse(lines[firstIdx]!.trim())
402
+ } catch {
403
+ return { body: stripped } // 首行不是 JSON → 兜底
404
+ }
405
+ if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
406
+ return { body: stripped }
407
+ }
408
+ const rec = parsed as { expectedDelta?: unknown; risk?: unknown }
409
+ if (!('expectedDelta' in rec)) return { body: stripped } // 不带 ε 的 JSON 不吃
410
+
411
+ const expectedEffect = typeof rec.expectedDelta === 'number' ? rec.expectedDelta : undefined
412
+ const risk = typeof rec.risk === 'string' ? rec.risk : undefined
413
+ // 剥掉 JSON 行本身 + 紧随其后的空行
414
+ const body = lines
415
+ .slice(firstIdx + 1)
416
+ .join('\n')
417
+ .replace(/^[ \t]*\n/, '')
418
+
419
+ return {
420
+ body,
421
+ ...(expectedEffect !== undefined ? { expectedEffect } : {}),
422
+ ...(risk !== undefined ? { risk } : {}),
423
+ }
424
+ }
425
+
369
426
  export async function generateProseContent(
370
427
  signal: CrsiSignal,
371
428
  llm: Llm,
372
429
  filePath: string,
373
430
  originalContent: string,
374
- ): Promise<string | null> {
431
+ ): Promise<ProsePrediction | null> {
375
432
  const prompt = buildGenerateProsePrompt(signal, filePath, originalContent)
376
433
  const response = await collectLlmText(llm, prompt)
377
434
  if (!response) return null
378
- return stripMarkdownFence(response)
435
+ return parseProsePrediction(response)
379
436
  }
380
437
 
381
438
  export interface ProseProposalResult {
@@ -383,6 +440,10 @@ export interface ProseProposalResult {
383
440
  newContent: string
384
441
  originalContent: string
385
442
  description: string
443
+ /** ε:由模型在正文之前写下(见 parseProsePrediction)。 */
444
+ expectedEffect?: number
445
+ /** R:风险声明。 */
446
+ risk?: string
386
447
  }
387
448
 
388
449
  export async function produceProseProposal(
@@ -401,10 +462,17 @@ export async function produceProseProposal(
401
462
  return null
402
463
  }
403
464
 
404
- const newContent = await generateProseContent(signal, llm, filePath, originalContent)
405
- if (!newContent) return null
465
+ const generated = await generateProseContent(signal, llm, filePath, originalContent)
466
+ if (!generated || !generated.body) return null
406
467
 
407
- return { filePath, newContent, originalContent, description: signal.title }
468
+ return {
469
+ filePath,
470
+ newContent: generated.body,
471
+ originalContent,
472
+ description: signal.title,
473
+ ...(generated.expectedEffect !== undefined ? { expectedEffect: generated.expectedEffect } : {}),
474
+ ...(generated.risk !== undefined ? { risk: generated.risk } : {}),
475
+ }
408
476
  }
409
477
 
410
478
  const SKILL_DIRS: Array<[string, string]> = [
@@ -582,6 +650,7 @@ export async function produceCrossoverProposal(
582
650
  newContent: string
583
651
  originalContent: string
584
652
  blastRadius: string[]
653
+ merge: boolean
585
654
  } | null> {
586
655
  const response = await collectLlmText(llm, buildCrossoverPrompt(currentLessons))
587
656
  if (!response) return null
@@ -607,5 +676,6 @@ export async function produceCrossoverProposal(
607
676
  newContent,
608
677
  originalContent: currentLessons,
609
678
  blastRadius: [LESSONS_FILE],
679
+ merge: true,
610
680
  }
611
681
  }
@@ -19,6 +19,7 @@ import { mkdirSync, rmSync, existsSync, writeFileSync, readFileSync, readdirSync
19
19
  import { join, resolve, sep, posix } from 'node:path'
20
20
  import { tmpdir, homedir } from 'node:os'
21
21
  import { randomUUID } from 'node:crypto'
22
+ import { LESSONS_FILE, MANAGED_RULES_FILE } from './crsi-producer'
22
23
 
23
24
  // ── Types ──
24
25
 
@@ -195,6 +196,70 @@ export function validateBlastRadius(proposal: {
195
196
  return null
196
197
  }
197
198
 
199
+ /**
200
+ * 脚手架三项计数(B_H 的度量)。按 filePath 分派语义单位:
201
+ * 教训段数 / 受管理规则条数 / 其余按 UTF-8 字节数。
202
+ *
203
+ * **为什么教训/规则文件不计字节**:合并会重写散文,字节数随措辞涨落。把字节计入,
204
+ * 会让「删二增一」因新写的合并段比原来两段更长而被误拦 —— 即闸会挡掉它本该允许的那件事。
205
+ * skill 文件没有可用的语义单位(它的「条数」就是文件本身),才退到字节数。
206
+ *
207
+ * `## ` 的口径与 `removeLessonSections` 逐字一致 —— 闸数的必须是 crossover 真正删得掉的那些
208
+ * 单位,否则两把尺子会各说各话。**刻意不声称与 `extractCrsiLessonSummaries` 一致**:后者用
209
+ * `/^##\s+(.+?)\s*$/`,还认 `##\t`,而 `startsWith('## ')` 不认(`'##\ta: 1'` 在此计 0、
210
+ * 在那里计 1)。闸依赖的是「删得掉」,故按前者对齐;这个差是选择,不是遗漏。
211
+ *
212
+ * 分派按**解析后**的路径(`resolve` 两侧同调,`cwd` 相消)—— 字面量比较时,`./` 前缀或
213
+ * 绝对形式的教训路径会静默落到**字节**分支,而那正是上面说绝不该用在教训文件上的那把尺子。
214
+ * 兄弟守卫 `isProtectedPath` 本身不做规范化(纯前缀比较);是调用点 `CrsiSandbox.applyModification`
215
+ * 先 `posix.normalize` 再调它(`proposal-guard.ts` 那条调用点未规范化)。
216
+ */
217
+ export function measureScaffold(
218
+ filePath: string,
219
+ content: string,
220
+ ): { lessons: number; rules: number; bytes: number } {
221
+ if (resolve(filePath) === resolve(LESSONS_FILE)) {
222
+ const lessons = content.split('\n').filter((l) => l.startsWith('## ')).length
223
+ return { lessons, rules: 0, bytes: 0 }
224
+ }
225
+ if (resolve(filePath) === resolve(MANAGED_RULES_FILE)) {
226
+ const rules = (content.match(/id: '/g) ?? []).length
227
+ return { lessons: 0, rules, bytes: 0 }
228
+ }
229
+ return { lessons: 0, rules: 0, bytes: Buffer.byteLength(content, 'utf-8') }
230
+ }
231
+
232
+ /**
233
+ * 合并型提案的收敛闸(B_H)。**零写死常数** —— 它不设上界,只要求「合并这件事本身别把
234
+ * 脚手架抬高」。RSIH 的 ‖H‖≤B_H 需要一个数,是因为它必须允许学到上界、到顶再强制合并;
235
+ * 本仓已有 dedup 那一半(教训按 `## category: title` 幂等、规则按 id 幂等),
236
+ * 缺的只是另一半,而那一半不需要数:**闸只在「试图整合」那一刻开火**。
237
+ *
238
+ * 返回拒绝理由,合法时返回 null(同 validateBlastRadius 的签名)。
239
+ */
240
+ export function validateMergeConvergence(proposal: {
241
+ filePath?: string
242
+ originalContent?: string
243
+ newContent?: string
244
+ merge?: boolean
245
+ }): string | null {
246
+ if (proposal.merge !== true) return null
247
+ // 无基线不是有增长:手工路径在文件不存在时正是这个形态(commands.ts 的宽松模式)。
248
+ if (!proposal.originalContent || !proposal.newContent) return null
249
+
250
+ const filePath = proposal.filePath ?? ''
251
+ const before = measureScaffold(filePath, proposal.originalContent)
252
+ const after = measureScaffold(filePath, proposal.newContent)
253
+
254
+ const rose: string[] = []
255
+ if (after.lessons > before.lessons) rose.push(`教训段数 ${before.lessons} → ${after.lessons}`)
256
+ if (after.rules > before.rules) rose.push(`规则条数 ${before.rules} → ${after.rules}`)
257
+ if (after.bytes > before.bytes) rose.push(`字节数 ${before.bytes} → ${after.bytes}`)
258
+ if (rose.length === 0) return null
259
+
260
+ return `合并型提案必须收敛,但脚手架增长了:${rose.join(';')}。`
261
+ }
262
+
198
263
  // ── Sandbox ──
199
264
 
200
265
  export class CrsiSandbox {
@@ -21,14 +21,21 @@ import { PreFlightChecker } from './preflight-checker'
21
21
  import { createDefaultPostFlightChecker } from './post-flight-checker'
22
22
  import { WorkingMemory } from './working-memory'
23
23
  import { RedTeam } from './red-team'
24
- import { isProtectedPath, validateBlastRadius, PROTECTED_CRITICAL_FILES } from './crsi-sandbox'
24
+ import {
25
+ isProtectedPath,
26
+ validateBlastRadius,
27
+ validateMergeConvergence,
28
+ PROTECTED_CRITICAL_FILES,
29
+ } from './crsi-sandbox'
25
30
  import {
26
31
  produceRuleProposal,
27
32
  MANAGED_RULES_FILE,
33
+ LESSONS_FILE,
28
34
  buildLessonContent,
29
35
  renderManagedRuleSource,
30
36
  } from './crsi-producer'
31
37
  import type { CrsiSignal } from './crsi-producer'
38
+ import { predictionHit } from './improvement-track'
32
39
  import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
33
40
 
34
41
  // ── Types ──
@@ -43,6 +50,12 @@ export interface EvalResult {
43
50
  detail?: string
44
51
  /** 契约角色。缺省 neutral。 */
45
52
  role?: ContractRole
53
+ /**
54
+ * anchor 契约的**定义处真源**。`role` 由此派生(见 runEval 末尾的回填)。
55
+ * `ANCHOR_CONTRACT_IDS` 退为**独立声明**,两者由 test/integrity/anchor-contract-wiring
56
+ * 守卫两向比对 —— 两处独立陈述同一件事,它们才可能不一致,守卫才有内容。
57
+ */
58
+ anchor?: true
46
59
  }
47
60
 
48
61
  export interface EvalReport {
@@ -70,6 +83,8 @@ export const ANCHOR_CONTRACT_IDS: ReadonlySet<string> = new Set([
70
83
  'red-team-zero-gaps',
71
84
  'producer-rule-shape',
72
85
  'producer-rule-idempotent',
86
+ 'prediction-hit-truth-table',
87
+ 'merge-convergence-gate',
73
88
  'self-report-diagnostic',
74
89
  ])
75
90
 
@@ -82,10 +97,24 @@ export function regressedAnchors(results: EvalResult[]): string[] {
82
97
 
83
98
  const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
84
99
 
100
+ /** 落盘的契约粒度投影 —— 只要 id/passed/role(EvalResult 的 description/detail 不落盘)。 */
101
+ export interface ContractResultRecord {
102
+ id: string
103
+ passed: boolean
104
+ role?: ContractRole
105
+ }
106
+
107
+ /** 一次评估的契约粒度快照:契约 id → 是否通过。 */
108
+ export type ContractSnapshot = Record<string, boolean>
109
+
110
+ function toContractResultRecord(r: EvalResult): ContractResultRecord {
111
+ return { id: r.id, passed: r.passed, ...(r.role ? { role: r.role } : {}) }
112
+ }
113
+
85
114
  /** 追加一次评估分数到 rewards 日志(按奖励函数名键控)。 */
86
115
  export function appendEvalScore(
87
116
  name: string,
88
- report: { score: number; passed: number; total: number },
117
+ report: { score: number; passed: number; total: number; results?: EvalResult[] },
89
118
  ): void {
90
119
  try {
91
120
  mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
@@ -97,6 +126,9 @@ export function appendEvalScore(
97
126
  score: report.score,
98
127
  passed: report.passed,
99
128
  total: report.total,
129
+ // 契约粒度(B1)。缺省不写该键:B1 之前落盘的旧记录没有它,
130
+ // 读取侧跳过 —— 这是向后兼容的承重判据,别改成 `results: []`。
131
+ ...(report.results ? { results: report.results.map(toContractResultRecord) } : {}),
100
132
  }) + '\n',
101
133
  'utf-8',
102
134
  )
@@ -120,6 +152,105 @@ export function getLastEvalScore(name: string): number | null {
120
152
  }
121
153
  }
122
154
 
155
+ /**
156
+ * 某奖励函数最近 n 次**按契约粒度**落盘的记录,新→旧。
157
+ *
158
+ * 只认带 `results` 的记录 —— B1 之前落盘的旧记录(只有聚合分数)被跳过,
159
+ * 于是调用方不必区分新旧形态。逐行容错:坏行跳过而不是让整条历史归零
160
+ * (`fixCache` 负责清理坏行)。
161
+ */
162
+ export function getContractHistory(name: string, n = 3): ContractSnapshot[] {
163
+ try {
164
+ if (!existsSync(SCORES_FILE)) return []
165
+ const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
166
+ const out: ContractSnapshot[] = []
167
+ for (let i = lines.length - 1; i >= 0 && out.length < n; i--) {
168
+ let rec: { name?: string; results?: ContractResultRecord[] }
169
+ try {
170
+ rec = JSON.parse(lines[i]!) as { name?: string; results?: ContractResultRecord[] }
171
+ } catch {
172
+ continue
173
+ }
174
+ if (rec.name !== name || !Array.isArray(rec.results)) continue
175
+ const snap: ContractSnapshot = {}
176
+ for (const r of rec.results) {
177
+ if (typeof r?.id === 'string') snap[r.id] = r.passed === true
178
+ }
179
+ out.push(snap)
180
+ }
181
+ return out
182
+ } catch {
183
+ return []
184
+ }
185
+ }
186
+
187
+ export type ContractDelta = 'regressed' | 'fixed' | 'flaky' | 'new' | 'gone'
188
+
189
+ /**
190
+ * 纯函数:当前 run vs 历史 → 每条契约的变化。**只报变化**,未变化的契约不出现。
191
+ *
192
+ * delta 判据(`history` 新→旧):
193
+ * - 历史上没出现过 → `new`
194
+ * - 历史上 true/false 都出现过 → `flaky`(压过 regressed/fixed:抖动的契约
195
+ * 不该被报成「已修复」或「回归」)
196
+ * - 上次与本次相反 → `regressed`(上次 PASS→本次 FAIL)/ `fixed`
197
+ * - 本次没有但历史有 → `gone`
198
+ *
199
+ * 为什么 `flaky` 要压过相邻两次的比较:只比相邻两次会把一个每次都在翻的契约
200
+ * 误报成「真回归」,而那正是这个账本要区分开的东西。
201
+ */
202
+ export function diffContractHistory(
203
+ current: ContractResultRecord[],
204
+ history: ContractSnapshot[],
205
+ ): { id: string; delta: ContractDelta; role?: ContractRole }[] {
206
+ const out: { id: string; delta: ContractDelta; role?: ContractRole }[] = []
207
+ const seen = new Set<string>()
208
+ for (const c of current) {
209
+ seen.add(c.id)
210
+ const past = history.filter((h) => c.id in h).map((h) => h[c.id] === true)
211
+ let delta: ContractDelta
212
+ if (past.length === 0) delta = 'new'
213
+ else if (past.includes(true) && past.includes(false)) delta = 'flaky'
214
+ else if (past[0] === !c.passed) delta = c.passed ? 'fixed' : 'regressed'
215
+ else continue // 未变化
216
+ out.push({ id: c.id, delta, ...(c.role ? { role: c.role } : {}) })
217
+ }
218
+ for (const h of history) {
219
+ for (const id of Object.keys(h)) {
220
+ if (seen.has(id)) continue
221
+ seen.add(id)
222
+ out.push({ id, delta: 'gone' })
223
+ }
224
+ }
225
+ return out
226
+ }
227
+
228
+ /**
229
+ * 把 diffContractHistory 的结果渲染成展示行。纯函数——不做 I/O、不读时钟。
230
+ * 返回空数组表示「无变化」,调用方据此决定是否打印标题。
231
+ */
232
+ export function renderContractDiff(
233
+ deltas: { id: string; delta: ContractDelta; role?: ContractRole }[],
234
+ ): string[] {
235
+ const text: Record<ContractDelta, string> = {
236
+ regressed: '上次 PASS,本次 FAIL(回归)',
237
+ fixed: '上次 FAIL,本次 PASS(已修复)',
238
+ flaky: '近几次结果不一致(抖动)',
239
+ new: '本次新增的契约',
240
+ gone: '本次未出现(已移出契约集)',
241
+ }
242
+ const icon: Record<ContractDelta, string> = {
243
+ regressed: '❌',
244
+ fixed: '✅',
245
+ flaky: '⚠️',
246
+ new: '🆕',
247
+ gone: '➖',
248
+ }
249
+ return deltas.map(
250
+ (d) => `${icon[d.delta]} ${d.id} ← ${text[d.delta]}${d.role ? ` \`${d.role}\`` : ''}`,
251
+ )
252
+ }
253
+
123
254
  // ── Harness ──
124
255
 
125
256
  /** 构建隔离组件,避免读用户 ~/.mipham 运行时状态。 */
@@ -147,6 +278,7 @@ export function runEval(): EvalReport {
147
278
  id: 'rule-timeout',
148
279
  description: '内置 timeout 规则命中低超时的 npm install',
149
280
  passed: timeout.modified.timeout === 300000,
281
+ anchor: true,
150
282
  })
151
283
 
152
284
  const gitForce = ruleEngine.intercept('Bash', {
@@ -157,6 +289,7 @@ export function runEval(): EvalReport {
157
289
  id: 'rule-git-force',
158
290
  description: 'git --force 触发告警',
159
291
  passed: gitForce.warnings.length > 0,
292
+ anchor: true,
160
293
  })
161
294
 
162
295
  const disabledRule: import('./rule-engine').ToolRule = {
@@ -174,6 +307,7 @@ export function runEval(): EvalReport {
174
307
  id: 'rule-disabled-skip',
175
308
  description: '禁用规则被跳过',
176
309
  passed: disabled.warnings.length === 0,
310
+ anchor: true,
177
311
  })
178
312
 
179
313
  // ── 宪法(ground truth:8 原则 + facet 映射 + 愿力序言) ──
@@ -182,6 +316,7 @@ export function runEval(): EvalReport {
182
316
  id: 'constitution-8-principles',
183
317
  description: '宪法含 8 条原则',
184
318
  passed: principles.length === 8,
319
+ anchor: true,
185
320
  })
186
321
 
187
322
  const prajna = principles.filter((p) => p.facet === 'prajna').length
@@ -191,12 +326,14 @@ export function runEval(): EvalReport {
191
326
  id: 'constitution-facets',
192
327
  description: 'facet 映射 智3 / 金刚5 / 悲0',
193
328
  passed: prajna === 3 && vajra === 5 && karuna === 0,
329
+ anchor: true,
194
330
  })
195
331
 
196
332
  results.push({
197
333
  id: 'constitution-preamble',
198
334
  description: '愿力序言已注入',
199
335
  passed: !!DEFAULT_CONSTITUTION.preamble && DEFAULT_CONSTITUTION.preamble.includes('悲'),
336
+ anchor: true,
200
337
  })
201
338
 
202
339
  // ── 沙箱只读边界(ground truth:受保护路径被拒) ──
@@ -206,7 +343,12 @@ export function runEval(): EvalReport {
206
343
  ['sandbox-protected-machinery', 'apps/cli/src/core/crsi-sandbox.ts'],
207
344
  ]
208
345
  for (const [id, path] of protectedChecks) {
209
- results.push({ id, description: `受保护路径被拒: ${path}`, passed: isProtectedPath(path) })
346
+ results.push({
347
+ id,
348
+ description: `受保护路径被拒: ${path}`,
349
+ passed: isProtectedPath(path),
350
+ anchor: true,
351
+ })
210
352
  }
211
353
 
212
354
  // ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
@@ -216,6 +358,7 @@ export function runEval(): EvalReport {
216
358
  description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
217
359
  passed: unprotected.length === 0,
218
360
  ...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
361
+ anchor: true,
219
362
  })
220
363
 
221
364
  // ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
@@ -226,6 +369,7 @@ export function runEval(): EvalReport {
226
369
  validateBlastRadius({ blastRadius: undefined }) !== null &&
227
370
  validateBlastRadius({ blastRadius: [] }) !== null &&
228
371
  validateBlastRadius({ blastRadius: ['apps/cli/src/foo.ts'] }) === null,
372
+ anchor: true,
229
373
  })
230
374
 
231
375
  // ── 安全(ground truth:16 攻击零漏过) ──
@@ -235,6 +379,7 @@ export function runEval(): EvalReport {
235
379
  description: '16 个对抗场景零漏过',
236
380
  passed: redTeam.passedThrough === 0,
237
381
  detail: `score=${redTeam.score}, passedThrough=${redTeam.passedThrough}, falsePositives=${redTeam.falsePositives}`,
382
+ anchor: true,
238
383
  })
239
384
 
240
385
  // ── producer 行为(ground truth:固化规则产出正确 shape + 幂等) ──
@@ -255,6 +400,7 @@ export function runEval(): EvalReport {
255
400
  ruleProposal.newContent.includes("source: 'managed'") &&
256
401
  ruleProposal.newContent.includes('timeout: 300000') &&
257
402
  ruleProposal.newContent.includes('enabled: true'),
403
+ anchor: true,
258
404
  })
259
405
 
260
406
  results.push({
@@ -262,6 +408,7 @@ export function runEval(): EvalReport {
262
408
  description: '同名规则重复产出被拒(幂等)',
263
409
  passed:
264
410
  ruleProposal !== null && produceRuleProposal(frozenSignal, ruleProposal.newContent) === null,
411
+ anchor: true,
265
412
  })
266
413
 
267
414
  // ── 组件归因(ground truth:缺省 experiential、显式组件透传、非 experiential 不进 managed-rule) ──
@@ -334,6 +481,46 @@ export function runEval(): EvalReport {
334
481
  results.push({ ...judgeBehaviorTask(task, ruleEngine), role: 'target' })
335
482
  }
336
483
 
484
+ // ── ε 预测命中真值表(ground truth:命中判据不叠加统计阈值) ──
485
+ // `(20, 20)` 那条**承重**:判据是 `deltaMean >= predicted` 而 `>=` 与 `>` 只在
486
+ // `predicted === deltaMean` 处分歧 ⇒ 少了它,「把 >= 翻成 >」在契约上不可观测。
487
+ results.push({
488
+ id: 'prediction-hit-truth-table',
489
+ description: 'predictionHit 真值表(返回值):未达不算、达到或恰好相等算命中、缺席恒 false',
490
+ passed:
491
+ predictionHit(50, 20) === false &&
492
+ predictionHit(10, 20) === true &&
493
+ predictionHit(20, 20) === true &&
494
+ predictionHit(undefined, 20) === false,
495
+ anchor: true,
496
+ })
497
+
498
+ // ── B_H 合并型收敛闸(ground truth:净增被拒、删二增一通过、非合并型不受此闸) ──
499
+ results.push({
500
+ id: 'merge-convergence-gate',
501
+ description: '合并型净增被拒、删二增一通过、merge=false 净增通过',
502
+ passed:
503
+ validateMergeConvergence({
504
+ filePath: LESSONS_FILE,
505
+ originalContent: '## a: 1\n\n## b: 2\n',
506
+ newContent: '## a: 1\n\n## b: 2\n\n## c: 3\n',
507
+ merge: true,
508
+ }) !== null &&
509
+ validateMergeConvergence({
510
+ filePath: LESSONS_FILE,
511
+ originalContent: '## a: 1\n\n## b: 2\n',
512
+ newContent: '## ab: merged\n',
513
+ merge: true,
514
+ }) === null &&
515
+ validateMergeConvergence({
516
+ filePath: LESSONS_FILE,
517
+ originalContent: '## a: 1\n',
518
+ newContent: '## a: 1\n\n## b: 2\n',
519
+ merge: false,
520
+ }) === null,
521
+ anchor: true,
522
+ })
523
+
337
524
  // ── 自报分数只作诊断:评分路径无 LLM,分数来自 ground-truth 契约而非模型自报 ──
338
525
  // anchor 锁死「评分组件不暴露 LLM 的 chat 能力」。4 个组件(ruleEngine/constitution/
339
526
  // errorDB/preflight)都是确定性组件(runEval 同步评分)。若未来有人把 LLM 注入评分
@@ -347,11 +534,13 @@ export function runEval(): EvalReport {
347
534
  id: 'self-report-diagnostic',
348
535
  description: '评分无 LLM:机制哨兵组件不暴露 chat 能力(分数只来自 ground-truth,非模型自报)',
349
536
  passed: !llmInjected,
537
+ anchor: true,
350
538
  })
351
539
 
352
- // 角色标注:anchor 走集中清单(门保护面单一真源),target 已在上方循环内联。
540
+ // 角色标注:anchor 由契约**定义处内联的标记**派生(真源),
541
+ // ANCHOR_CONTRACT_IDS 退为独立声明 —— 两者由 anchor-contract-wiring 守卫两向比对。
353
542
  for (const r of results) {
354
- if (ANCHOR_CONTRACT_IDS.has(r.id)) r.role = 'anchor'
543
+ if (r.anchor) r.role = 'anchor'
355
544
  }
356
545
 
357
546
  // anchor 自检(ground truth:所有 anchor 契约必须全绿,否则门拒)。
@@ -22,6 +22,17 @@ export interface ImprovementReport {
22
22
  noise: number
23
23
  minEffect: number
24
24
  verdict: ImprovementVerdict
25
+ /** ε:提交者事前写下的预期提升点数。缺席 = 该记录没有预登记。 */
26
+ predictedDelta?: number
27
+ /** 预测是否命中。与 predictedDelta 同时出现、同时缺席(JSON 序列化会丢掉 undefined 键)。 */
28
+ predictionHit?: boolean
29
+ /**
30
+ * B2 代价维:与分数数组逐项对齐的前/后耗时。**只记录,不进任何门禁** ——
31
+ * `verdict` / `deltaMean` / `minEffect` 一律不看这两个字段。
32
+ * 缺席(而非空数组)= 该记录早于代价维落地,或该样本未记代价。
33
+ */
34
+ baselineDurations?: number[]
35
+ postDurations?: number[]
25
36
  }
26
37
 
27
38
  export interface ImprovementRecord extends ImprovementReport {
@@ -55,6 +66,7 @@ function stdDev(xs: number[]): number {
55
66
  export function buildImprovementReport(
56
67
  sample: SkillDeltaSample,
57
68
  changeSet: string[],
69
+ predicted?: number,
58
70
  ): ImprovementReport {
59
71
  const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
60
72
  const noise = stdDev(sample.baselineScores)
@@ -70,6 +82,19 @@ export function buildImprovementReport(
70
82
  noise,
71
83
  minEffect,
72
84
  verdict,
85
+ // 两个字段同生同灭:缺席预测必须**键不存在**,而不是 predictionHit: false。
86
+ // 理由**不是**「predictionHit: false 会被算进分母」—— 分母只认 `predictedDelta !== undefined`
87
+ //(只写 `predictionHit: false` 而不写 `predictedDelta` 会被整条忽略)。真正的风险是**反过来的半条**:
88
+ // 有 `predictedDelta` 而无 `predictionHit` ⇒ 该条进了分母,却永远不可能被算成命中
89
+ //(分子只认 `predictionHit === true`),等于一条静默的「未命中」。
90
+ ...(predicted !== undefined
91
+ ? { predictedDelta: predicted, predictionHit: predictionHit(predicted, deltaMean) }
92
+ : {}),
93
+ // 代价维:**两条同生同灭**,与上面 ε 那条同理 —— 只写一半(有 baselineDurations
94
+ // 而无 postDurations)会让「代价」这件事在记录里既非有也非无,读侧无从判断。
95
+ ...(sample.baselineDurations !== undefined && sample.postDurations !== undefined
96
+ ? { baselineDurations: sample.baselineDurations, postDurations: sample.postDurations }
97
+ : {}),
73
98
  }
74
99
  }
75
100
 
@@ -107,6 +132,52 @@ export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
107
132
  return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
108
133
  }
109
134
 
135
+ /**
136
+ * 预测命中:事前写下的点数被实际达到。缺席预测(undefined)不计入。
137
+ *
138
+ * 刻意**不叠加 `minEffect`** —— ε 是提交者自己写下的数,判据就是「达到没达到」;
139
+ * 再套一层统计阈值会让两个数打架,且使「命中」不可复算。
140
+ */
141
+ export function predictionHit(predicted: number | undefined, deltaMean: number): boolean {
142
+ return predicted !== undefined && deltaMean >= predicted
143
+ }
144
+
145
+ /**
146
+ * ε 命中率。分母 = **有预测的记录数**(判据取 `predictedDelta !== undefined`),
147
+ * 复用既有 wilsonInterval。无预测的记录既不入分子也不入分母。
148
+ */
149
+ export function predictionHitRate(records: ImprovementRecord[]): {
150
+ total: number
151
+ hits: number
152
+ rate: number
153
+ lo: number
154
+ hi: number
155
+ } {
156
+ const judged = records.filter((r) => r.predictedDelta !== undefined)
157
+ const total = judged.length
158
+ const hits = judged.filter((r) => r.predictionHit === true).length
159
+ const { lo, hi } = wilsonInterval(hits, total)
160
+ return { total, hits, rate: total === 0 ? 0 : hits / total, lo, hi }
161
+ }
162
+
163
+ /**
164
+ * 代价维的只读展示(B2):前/后均值一行。两个耗时数组缺席 → null,调用方据此整行不打印。
165
+ *
166
+ * **刻意不下结论**:倍数只是描述,不是判据。把「代价过高」变成 verdict 的一部分需要
167
+ * 样本量支撑,而当前 k 默认 3、`minEffect` 已经要 `max(20, 2×噪声)` —— 再塞一个维度
168
+ * 只会把统计问题变得更糟。所以这里只回答「花了多少」,不回答「值不值」。
169
+ *
170
+ * 基线均值为 0 时**不给倍数**:那个比值是 ∞(或 0/0 的 NaN),打出来是假读数。
171
+ */
172
+ export function formatCostLine(report: ImprovementReport): string | null {
173
+ const { baselineDurations: before_, postDurations: after_ } = report
174
+ if (before_ === undefined || after_ === undefined) return null
175
+ const before = Math.round(mean(before_))
176
+ const after = Math.round(mean(after_))
177
+ const line = `⏱️ 代价: 均值 ${before}ms → ${after}ms`
178
+ return before > 0 ? `${line}(×${(after / before).toFixed(1)})` : line
179
+ }
180
+
110
181
  // ── 台账 ──
111
182
 
112
183
  export function improvementPath(): string {
@@ -68,6 +68,8 @@ export interface TaskPerformanceReport {
68
68
  score: number
69
69
  results: TaskPerformanceResult[]
70
70
  failures: string[]
71
+ /** B2 代价维:整轮(LLM 生成 + 冻结测试判定)的墙钟耗时。只记录,不参与任何判定。 */
72
+ durationMs: number
71
73
  }
72
74
 
73
75
  /** 剥掉 LLM 可能包裹的 markdown 代码块(```ts ... ```),拿到裸代码。 */
@@ -101,6 +103,7 @@ export async function runTaskPerformance(
101
103
  const wanted = opts?.skill?.name
102
104
  const tasks = loadPerformanceTasks().filter((t) => (t.skill ?? undefined) === wanted)
103
105
  const results: TaskPerformanceResult[] = []
106
+ const startedAt = Date.now()
104
107
  for (const task of tasks) {
105
108
  const code = await collectGeneratedCode(llm, task.prompt, opts?.skill?.text)
106
109
  if (!code) {
@@ -127,6 +130,7 @@ export async function runTaskPerformance(
127
130
  score: results.length > 0 ? Math.round((passed / results.length) * 100) : 100,
128
131
  results,
129
132
  failures: results.filter((r) => !r.passed).map((r) => r.id),
133
+ durationMs: Date.now() - startedAt,
130
134
  }
131
135
  }
132
136
 
@@ -210,6 +214,13 @@ export interface SkillDeltaSample {
210
214
  skillName: string
211
215
  baselineScores: number[]
212
216
  postScores: number[]
217
+ /**
218
+ * B2 代价维:与分数数组**逐项对齐**的耗时(第 i 项就是产出第 i 个分数的那次采样)。
219
+ * 缺席 = 该样本未记代价(旧记录 / 旧调用点),读侧须按「缺席」处理、不得当成 0。
220
+ * 只记录,不参与改进判定 —— 样本量 k=3 时再加一个维度只会让统计问题更糟。
221
+ */
222
+ baselineDurations?: number[]
223
+ postDurations?: number[]
213
224
  }
214
225
 
215
226
  /**
@@ -227,18 +238,28 @@ export async function measureSkillDeltaRepeated(
227
238
  const k = opts?.k ?? 3
228
239
  const baselineScores: number[] = []
229
240
  const postScores: number[] = []
241
+ const baselineDurations: number[] = []
242
+ const postDurations: number[] = []
230
243
  for (let i = 0; i < k; i++) {
231
244
  const r = await runTaskPerformance(llm, {
232
245
  skill: { name: resolved.skillName, text: resolved.baselineText },
233
246
  })
234
247
  baselineScores.push(r.score)
248
+ baselineDurations.push(r.durationMs)
235
249
  }
236
250
  for (let i = 0; i < k; i++) {
237
251
  const r = await runTaskPerformance(llm, {
238
252
  skill: { name: resolved.skillName, text: resolved.postText },
239
253
  })
240
254
  postScores.push(r.score)
255
+ postDurations.push(r.durationMs)
241
256
  }
242
257
 
243
- return { skillName: resolved.skillName, baselineScores, postScores }
258
+ return {
259
+ skillName: resolved.skillName,
260
+ baselineScores,
261
+ postScores,
262
+ baselineDurations,
263
+ postDurations,
264
+ }
244
265
  }
@@ -9,7 +9,7 @@
9
9
  export const PACKAGE_NAME = '@miphamai/cli' as const
10
10
 
11
11
  /** 当前发布版本 */
12
- export const PACKAGE_VERSION = '0.81.9' as const
12
+ export const PACKAGE_VERSION = '0.83.0' as const
13
13
 
14
14
  /** npm install 全局安装命令 */
15
15
  export const NPM_INSTALL_COMMAND = `npm install -g ${PACKAGE_NAME}` as const
@@ -37,7 +37,13 @@ import {
37
37
  MANAGED_RULES_FILE,
38
38
  } from '../core/crsi-producer'
39
39
  import { prefilterProposal } from '../core/proposal-guard'
40
- import { runEval, appendEvalScore } from '../core/eval-harness'
40
+ import {
41
+ runEval,
42
+ appendEvalScore,
43
+ getContractHistory,
44
+ diffContractHistory,
45
+ renderContractDiff,
46
+ } from '../core/eval-harness'
41
47
  import { listRewardFns } from '../core/reward-fn'
42
48
  import {
43
49
  runTaskPerformance,
@@ -47,9 +53,11 @@ import {
47
53
  import { randomUUID } from 'node:crypto'
48
54
  import {
49
55
  buildImprovementReport,
56
+ formatCostLine,
50
57
  appendImprovement,
51
58
  readImprovements,
52
59
  improvementRate,
60
+ predictionHitRate,
53
61
  setPendingVerdict,
54
62
  getPendingVerdict,
55
63
  shouldBlockApproval,
@@ -804,6 +812,25 @@ const crsiStatsCmd: CommandHandler = async (ctx) => {
804
812
  }
805
813
  }
806
814
 
815
+ // ── ε 预测命中(prose 路径) ──
816
+ // 作废条款:样本不足就不下结论;台账攒够 20 条而判定样本仍 < 5 ⇒ 明写机制失效。
817
+ // 只打印、不写台账:本命令是只读的,而该结论每次都能从同一份 improvements.jsonl 重算出来。
818
+ const records = readImprovements()
819
+ const pred = predictionHitRate(records)
820
+ lines.push('')
821
+ lines.push('### ε 预测命中(prose 路径)')
822
+ if (pred.total < 5) {
823
+ lines.push(`样本不足(判定记录 ${pred.total} 条,需 ≥ 5)—— 不下结论。`)
824
+ if (records.length >= 20) {
825
+ lines.push('⚠️ ε 机制失效:prose 路径使用率过低(记录总数已达 20 而判定样本仍 < 5)。')
826
+ }
827
+ } else {
828
+ lines.push(
829
+ `命中率: ${pred.hits}/${pred.total} (${(pred.rate * 100).toFixed(0)}%, ` +
830
+ `Wilson 95% [${(pred.lo * 100).toFixed(0)}%, ${(pred.hi * 100).toFixed(0)}%])`,
831
+ )
832
+ }
833
+
807
834
  return { content: lines.join('\n') }
808
835
  }
809
836
 
@@ -881,9 +908,12 @@ const crsiModifyCmd: CommandHandler = async (ctx, args) => {
881
908
  ? 'regressed ⚠️'
882
909
  : 'inconclusive'
883
910
  const sign = report.deltaMean >= 0 ? '+' : ''
911
+ // B2 代价维:只展示、不进判定(`formatCostLine` 缺席时返回 null ⇒ 整行不打)。
912
+ const costLine = formatCostLine(report)
884
913
  improvementLine =
885
914
  `\n📊 改进判定: ${label} (delta ${sign}${report.deltaMean.toFixed(1)}, 噪声 ${report.noise.toFixed(1)}, 阈值 ${report.minEffect.toFixed(1)})` +
886
915
  `\n 改进率: ${rate.improved}/${rate.total} (${(rate.rate * 100).toFixed(0)}%, Wilson 95% [${(rate.lo * 100).toFixed(0)}%, ${(rate.hi * 100).toFixed(0)}%])` +
916
+ (costLine ? `\n${costLine}` : '') +
887
917
  (report.verdict === 'regressed' ? '\n ⚠️ 任务表现倒退:--approve 将被拒绝。' : '')
888
918
  }
889
919
  } catch {
@@ -970,11 +1000,40 @@ const crsiProposeCmd: CommandHandler = async (ctx, args) => {
970
1000
  return { content: `❌ 生成失败(phase: ${result.phase})。\n${result.error ?? ''}` }
971
1001
  }
972
1002
 
1003
+ // ε 预登记落地:prose 路径此前**不测量**(measureSkillDeltaRepeated 全仓库只有手工路径一个
1004
+ // 调用点)⇒ ε 曾在 A 流程登记、判定侧在 B 流程,两端永不相遇。这里补上测量。
1005
+ // 成本:每次提案多 6 次 LLM 调用(同手工路径 :867 的注释)。
1006
+ let predictionLine = ''
1007
+ try {
1008
+ const sample = await measureSkillDeltaRepeated(llm, {
1009
+ filePath: proposal.filePath,
1010
+ originalContent: proposal.originalContent,
1011
+ newContent: proposal.newContent,
1012
+ })
1013
+ if (sample) {
1014
+ const report = buildImprovementReport(sample, [proposal.filePath], proposal.expectedEffect)
1015
+ setPendingVerdict(report.verdict)
1016
+ appendImprovement({ ...report, id: randomUUID(), timestamp: new Date().toISOString() })
1017
+ if (report.predictionHit !== undefined) {
1018
+ predictionLine =
1019
+ `\n🎯 ε 预测命中: ${report.predictionHit ? '命中 ✅' : '未命中 ⚠️'}` +
1020
+ `(预测 ${report.predictedDelta},实际 delta ${report.deltaMean.toFixed(1)})`
1021
+ }
1022
+ // B2 代价维:**与手工路径同一行读数**,两条渲染路径都接(只接一条即本仓库记过的
1023
+ // 「局部正确全局遗漏」)。ε 行可以有、代价行可以没有,故各自独立追加。
1024
+ const costLine = formatCostLine(report)
1025
+ if (costLine) predictionLine += `\n${costLine}`
1026
+ }
1027
+ } catch {
1028
+ // 测量失败(LLM 不可用等)不阻断提案流程 —— 与手工路径 :889 的处置一致。
1029
+ }
1030
+
973
1031
  appendProseProposal({ id, filePath: proposal.filePath, timestamp: new Date().toISOString() })
974
1032
 
975
1033
  return {
976
1034
  content:
977
1035
  `✅ 已生成散文提议并跑过测试。审阅 diff:\n\n${result.diff}\n\n` +
1036
+ predictionLine +
978
1037
  '/crsi modify --approve 合并 | /crsi modify --reject 丢弃',
979
1038
  }
980
1039
  }
@@ -1087,14 +1146,26 @@ const crsiEvalCmd: CommandHandler = async (ctx, args) => {
1087
1146
  content: `❌ 未知 reward: ${rewardName}。可用: ${fns.map((f) => f.name).join(', ')}`,
1088
1147
  }
1089
1148
  }
1149
+ // 先读后写:比较基准是「上一次已落盘的态」,不依赖刚写进去那条排第几。
1150
+ const prev = getContractHistory(fn.name)
1090
1151
  const report = await fn.evaluate()
1152
+ const deltaLines = report.results
1153
+ ? renderContractDiff(diffContractHistory(report.results, prev))
1154
+ : []
1091
1155
  appendEvalScore(fn.name, report)
1092
1156
  return {
1093
- content: `得分 **${report.score}/100** (${report.passed}/${report.total})\n失败: ${report.failures.join(', ') || '无'}`,
1157
+ content: [
1158
+ `得分 **${report.score}/100** (${report.passed}/${report.total})`,
1159
+ `失败: ${report.failures.join(', ') || '无'}`,
1160
+ ...(deltaLines.length > 0 ? ['', '### 与上次相比', ...deltaLines] : []),
1161
+ ].join('\n'),
1094
1162
  }
1095
1163
  }
1096
1164
 
1165
+ // 先读后写(同上):基准是上一次落盘的态。
1166
+ const prevSnapshot = getContractHistory('mechanism-sentinel')
1097
1167
  const report = runEval()
1168
+ const deltaLines = renderContractDiff(diffContractHistory(report.results, prevSnapshot))
1098
1169
  appendEvalScore('mechanism-sentinel', report)
1099
1170
 
1100
1171
  const lines: string[] = ['## 🧪 CRSI Eval Harness', '']
@@ -1109,6 +1180,11 @@ const crsiEvalCmd: CommandHandler = async (ctx, args) => {
1109
1180
  if (report.failures.length > 0) {
1110
1181
  lines.push('', `❌ 失败任务: ${report.failures.join(', ')}`)
1111
1182
  }
1183
+ // 只读展示:账本按契约粒度落盘后,才能回答「是哪条契约翻的」。
1184
+ // 这里不做任何自动决策——闸门仍只看 regressedAnchors / score。
1185
+ if (deltaLines.length > 0) {
1186
+ lines.push('', '### 与上次相比', ...deltaLines)
1187
+ }
1112
1188
 
1113
1189
  // 奖励函数注册表(reward function = policy→feedback 抽象可见)
1114
1190
  // 传 llm 列出完整注册表(task-performance 需 llm 才能跑,但构造它零 LLM 调用)。