@miphamai/cli 0.81.9 → 0.83.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/crsi-modify.ts +34 -1
- package/src/core/crsi-producer.ts +77 -7
- package/src/core/crsi-sandbox.ts +65 -0
- package/src/core/eval-harness.ts +194 -5
- package/src/core/improvement-track.ts +71 -0
- package/src/core/task-performance.ts +22 -1
- package/src/shared/package-info.ts +1 -1
- package/src/ui/commands.ts +78 -2
package/package.json
CHANGED
package/src/core/crsi-modify.ts
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
|
-
import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
|
|
16
|
+
import { CrsiSandbox, validateBlastRadius, validateMergeConvergence } from './crsi-sandbox'
|
|
17
17
|
import type { CrsiModificationResult } from './crsi-sandbox'
|
|
18
18
|
import { appendEvalScore, getLastEvalScore, regressedAnchors } from './eval-harness'
|
|
19
19
|
import { mechanismSentinel, type RewardFn } from './reward-fn'
|
|
@@ -36,6 +36,19 @@ export interface CrsiProposal {
|
|
|
36
36
|
* 局部正确、全局遗漏。自修改前必须摸清并声明全部受影响路径,否则 fail-closed 拒绝。
|
|
37
37
|
*/
|
|
38
38
|
blastRadius?: string[]
|
|
39
|
+
/**
|
|
40
|
+
* ε:提交者**事前**写下的预期效果(任务表现提升点数)。
|
|
41
|
+
* 缺席 = 不预测。判定见 improvement-track 的 predictionHit。
|
|
42
|
+
*/
|
|
43
|
+
expectedEffect?: number
|
|
44
|
+
/** R:风险声明(这次改动可能在哪方面变差)。缺席 = 未声明。 */
|
|
45
|
+
risk?: string
|
|
46
|
+
/**
|
|
47
|
+
* 声明这是一次**合并型**提案(整合已有内容,而非新增)。
|
|
48
|
+
* 只有它为 true 时 `validateMergeConvergence` 才开火 —— 学习本身就是增长,
|
|
49
|
+
* 对新增型设非增长约束等于永久禁掉 `/crsi propose`。
|
|
50
|
+
*/
|
|
51
|
+
merge?: boolean
|
|
39
52
|
}
|
|
40
53
|
|
|
41
54
|
// ── Pending proposal registry (两阶段闸门) ──
|
|
@@ -70,6 +83,26 @@ export async function runCrsiModification(
|
|
|
70
83
|
}
|
|
71
84
|
}
|
|
72
85
|
|
|
86
|
+
// B_H 收敛闸:合并型提案不得抬高脚手架成本。
|
|
87
|
+
// 位置与 blast radius 闸同序 —— 都在 worktree 之前,纯字符串比较、零磁盘 I/O、零副作用。
|
|
88
|
+
const convergenceError = validateMergeConvergence(proposal)
|
|
89
|
+
if (convergenceError) {
|
|
90
|
+
return {
|
|
91
|
+
modification: {
|
|
92
|
+
id: 'crsi-mod-rejected-merge-convergence',
|
|
93
|
+
description: proposal.description,
|
|
94
|
+
filePath: proposal.filePath,
|
|
95
|
+
newContent: proposal.newContent,
|
|
96
|
+
originalContent: proposal.originalContent ?? '',
|
|
97
|
+
crsiInsightId: proposal.crsiInsightId,
|
|
98
|
+
timestamp: new Date().toISOString(),
|
|
99
|
+
},
|
|
100
|
+
applied: false,
|
|
101
|
+
phase: 'failed',
|
|
102
|
+
error: convergenceError,
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
73
106
|
sandbox.createWorktree()
|
|
74
107
|
|
|
75
108
|
const applied = sandbox.applyModification({
|
|
@@ -336,7 +336,7 @@ export async function selectTargetSkill(
|
|
|
336
336
|
return extractFilePath(response, skillFiles)
|
|
337
337
|
}
|
|
338
338
|
|
|
339
|
-
const PROSE_GENERATE_PROMPT_VERSION = '1.
|
|
339
|
+
const PROSE_GENERATE_PROMPT_VERSION = '1.1.0'
|
|
340
340
|
|
|
341
341
|
function buildGenerateProsePrompt(
|
|
342
342
|
signal: CrsiSignal,
|
|
@@ -357,7 +357,12 @@ function buildGenerateProsePrompt(
|
|
|
357
357
|
'当前内容:',
|
|
358
358
|
originalContent,
|
|
359
359
|
'',
|
|
360
|
-
'
|
|
360
|
+
'返回格式(严格遵守,两段):',
|
|
361
|
+
'第 1 行:一行 JSON,写下你对这次改动的**预期效果**与**风险**:',
|
|
362
|
+
'{"expectedDelta": <number 或 null>, "risk": "<字符串>"}',
|
|
363
|
+
'- expectedDelta 是预期该 skill 的任务表现提升**点数**(可正可负;无法预测写 null)。',
|
|
364
|
+
'- risk 是这次改动可能在哪方面变差(一句话)。',
|
|
365
|
+
'第 2 行起:改进后的完整 markdown(保持 YAML frontmatter 的 name/description 字段,正文针对失败信号做针对性改进)。不要用代码围栏包住。',
|
|
361
366
|
].join('\n')
|
|
362
367
|
}
|
|
363
368
|
|
|
@@ -366,16 +371,68 @@ function stripMarkdownFence(text: string): string {
|
|
|
366
371
|
return match ? match[1]! : text
|
|
367
372
|
}
|
|
368
373
|
|
|
374
|
+
/** prose 提议的解析产物:正文 + 可选的事前预登记(ε 与风险声明)。 */
|
|
375
|
+
export interface ProsePrediction {
|
|
376
|
+
body: string
|
|
377
|
+
/** ε:事前写下的预期提升点数。缺席 = 模型没预测(含显式写 null)。 */
|
|
378
|
+
expectedEffect?: number
|
|
379
|
+
/** R:风险声明。缺席 = 未声明。 */
|
|
380
|
+
risk?: string
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* 解析 prose 响应:可选的一行 JSON 前缀(ε)+ 正文。
|
|
385
|
+
*
|
|
386
|
+
* 顺序是**先归一化、后嗅探**(不可颠倒):stripMarkdownFence 的正则锚在串首
|
|
387
|
+
* (/^```(?:markdown|md)?\s*\n…\n```\s*$/)。若先剥「首行围栏」再嗅探,正文尾部的
|
|
388
|
+
* 那个 ``` 就再没有东西去剥它 ⇒ 孤立的尾部围栏会进入写盘路径。
|
|
389
|
+
*
|
|
390
|
+
* 认领标记是**含 `expectedDelta` 键**(盖住 number 与显式 null 两种写法);
|
|
391
|
+
* 其余任何情况都走兜底 —— 正文 = 归一化后的原文,一字不改。
|
|
392
|
+
*/
|
|
393
|
+
export function parseProsePrediction(raw: string): ProsePrediction {
|
|
394
|
+
const stripped = stripMarkdownFence(raw)
|
|
395
|
+
const lines = stripped.split('\n')
|
|
396
|
+
const firstIdx = lines.findIndex((l) => l.trim() !== '')
|
|
397
|
+
if (firstIdx === -1) return { body: stripped }
|
|
398
|
+
|
|
399
|
+
let parsed: unknown
|
|
400
|
+
try {
|
|
401
|
+
parsed = JSON.parse(lines[firstIdx]!.trim())
|
|
402
|
+
} catch {
|
|
403
|
+
return { body: stripped } // 首行不是 JSON → 兜底
|
|
404
|
+
}
|
|
405
|
+
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
|
|
406
|
+
return { body: stripped }
|
|
407
|
+
}
|
|
408
|
+
const rec = parsed as { expectedDelta?: unknown; risk?: unknown }
|
|
409
|
+
if (!('expectedDelta' in rec)) return { body: stripped } // 不带 ε 的 JSON 不吃
|
|
410
|
+
|
|
411
|
+
const expectedEffect = typeof rec.expectedDelta === 'number' ? rec.expectedDelta : undefined
|
|
412
|
+
const risk = typeof rec.risk === 'string' ? rec.risk : undefined
|
|
413
|
+
// 剥掉 JSON 行本身 + 紧随其后的空行
|
|
414
|
+
const body = lines
|
|
415
|
+
.slice(firstIdx + 1)
|
|
416
|
+
.join('\n')
|
|
417
|
+
.replace(/^[ \t]*\n/, '')
|
|
418
|
+
|
|
419
|
+
return {
|
|
420
|
+
body,
|
|
421
|
+
...(expectedEffect !== undefined ? { expectedEffect } : {}),
|
|
422
|
+
...(risk !== undefined ? { risk } : {}),
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
|
|
369
426
|
export async function generateProseContent(
|
|
370
427
|
signal: CrsiSignal,
|
|
371
428
|
llm: Llm,
|
|
372
429
|
filePath: string,
|
|
373
430
|
originalContent: string,
|
|
374
|
-
): Promise<
|
|
431
|
+
): Promise<ProsePrediction | null> {
|
|
375
432
|
const prompt = buildGenerateProsePrompt(signal, filePath, originalContent)
|
|
376
433
|
const response = await collectLlmText(llm, prompt)
|
|
377
434
|
if (!response) return null
|
|
378
|
-
return
|
|
435
|
+
return parseProsePrediction(response)
|
|
379
436
|
}
|
|
380
437
|
|
|
381
438
|
export interface ProseProposalResult {
|
|
@@ -383,6 +440,10 @@ export interface ProseProposalResult {
|
|
|
383
440
|
newContent: string
|
|
384
441
|
originalContent: string
|
|
385
442
|
description: string
|
|
443
|
+
/** ε:由模型在正文之前写下(见 parseProsePrediction)。 */
|
|
444
|
+
expectedEffect?: number
|
|
445
|
+
/** R:风险声明。 */
|
|
446
|
+
risk?: string
|
|
386
447
|
}
|
|
387
448
|
|
|
388
449
|
export async function produceProseProposal(
|
|
@@ -401,10 +462,17 @@ export async function produceProseProposal(
|
|
|
401
462
|
return null
|
|
402
463
|
}
|
|
403
464
|
|
|
404
|
-
const
|
|
405
|
-
if (!
|
|
465
|
+
const generated = await generateProseContent(signal, llm, filePath, originalContent)
|
|
466
|
+
if (!generated || !generated.body) return null
|
|
406
467
|
|
|
407
|
-
return {
|
|
468
|
+
return {
|
|
469
|
+
filePath,
|
|
470
|
+
newContent: generated.body,
|
|
471
|
+
originalContent,
|
|
472
|
+
description: signal.title,
|
|
473
|
+
...(generated.expectedEffect !== undefined ? { expectedEffect: generated.expectedEffect } : {}),
|
|
474
|
+
...(generated.risk !== undefined ? { risk: generated.risk } : {}),
|
|
475
|
+
}
|
|
408
476
|
}
|
|
409
477
|
|
|
410
478
|
const SKILL_DIRS: Array<[string, string]> = [
|
|
@@ -582,6 +650,7 @@ export async function produceCrossoverProposal(
|
|
|
582
650
|
newContent: string
|
|
583
651
|
originalContent: string
|
|
584
652
|
blastRadius: string[]
|
|
653
|
+
merge: boolean
|
|
585
654
|
} | null> {
|
|
586
655
|
const response = await collectLlmText(llm, buildCrossoverPrompt(currentLessons))
|
|
587
656
|
if (!response) return null
|
|
@@ -607,5 +676,6 @@ export async function produceCrossoverProposal(
|
|
|
607
676
|
newContent,
|
|
608
677
|
originalContent: currentLessons,
|
|
609
678
|
blastRadius: [LESSONS_FILE],
|
|
679
|
+
merge: true,
|
|
610
680
|
}
|
|
611
681
|
}
|
package/src/core/crsi-sandbox.ts
CHANGED
|
@@ -19,6 +19,7 @@ import { mkdirSync, rmSync, existsSync, writeFileSync, readFileSync, readdirSync
|
|
|
19
19
|
import { join, resolve, sep, posix } from 'node:path'
|
|
20
20
|
import { tmpdir, homedir } from 'node:os'
|
|
21
21
|
import { randomUUID } from 'node:crypto'
|
|
22
|
+
import { LESSONS_FILE, MANAGED_RULES_FILE } from './crsi-producer'
|
|
22
23
|
|
|
23
24
|
// ── Types ──
|
|
24
25
|
|
|
@@ -195,6 +196,70 @@ export function validateBlastRadius(proposal: {
|
|
|
195
196
|
return null
|
|
196
197
|
}
|
|
197
198
|
|
|
199
|
+
/**
|
|
200
|
+
* 脚手架三项计数(B_H 的度量)。按 filePath 分派语义单位:
|
|
201
|
+
* 教训段数 / 受管理规则条数 / 其余按 UTF-8 字节数。
|
|
202
|
+
*
|
|
203
|
+
* **为什么教训/规则文件不计字节**:合并会重写散文,字节数随措辞涨落。把字节计入,
|
|
204
|
+
* 会让「删二增一」因新写的合并段比原来两段更长而被误拦 —— 即闸会挡掉它本该允许的那件事。
|
|
205
|
+
* skill 文件没有可用的语义单位(它的「条数」就是文件本身),才退到字节数。
|
|
206
|
+
*
|
|
207
|
+
* `## ` 的口径与 `removeLessonSections` 逐字一致 —— 闸数的必须是 crossover 真正删得掉的那些
|
|
208
|
+
* 单位,否则两把尺子会各说各话。**刻意不声称与 `extractCrsiLessonSummaries` 一致**:后者用
|
|
209
|
+
* `/^##\s+(.+?)\s*$/`,还认 `##\t`,而 `startsWith('## ')` 不认(`'##\ta: 1'` 在此计 0、
|
|
210
|
+
* 在那里计 1)。闸依赖的是「删得掉」,故按前者对齐;这个差是选择,不是遗漏。
|
|
211
|
+
*
|
|
212
|
+
* 分派按**解析后**的路径(`resolve` 两侧同调,`cwd` 相消)—— 字面量比较时,`./` 前缀或
|
|
213
|
+
* 绝对形式的教训路径会静默落到**字节**分支,而那正是上面说绝不该用在教训文件上的那把尺子。
|
|
214
|
+
* 兄弟守卫 `isProtectedPath` 本身不做规范化(纯前缀比较);是调用点 `CrsiSandbox.applyModification`
|
|
215
|
+
* 先 `posix.normalize` 再调它(`proposal-guard.ts` 那条调用点未规范化)。
|
|
216
|
+
*/
|
|
217
|
+
export function measureScaffold(
|
|
218
|
+
filePath: string,
|
|
219
|
+
content: string,
|
|
220
|
+
): { lessons: number; rules: number; bytes: number } {
|
|
221
|
+
if (resolve(filePath) === resolve(LESSONS_FILE)) {
|
|
222
|
+
const lessons = content.split('\n').filter((l) => l.startsWith('## ')).length
|
|
223
|
+
return { lessons, rules: 0, bytes: 0 }
|
|
224
|
+
}
|
|
225
|
+
if (resolve(filePath) === resolve(MANAGED_RULES_FILE)) {
|
|
226
|
+
const rules = (content.match(/id: '/g) ?? []).length
|
|
227
|
+
return { lessons: 0, rules, bytes: 0 }
|
|
228
|
+
}
|
|
229
|
+
return { lessons: 0, rules: 0, bytes: Buffer.byteLength(content, 'utf-8') }
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* 合并型提案的收敛闸(B_H)。**零写死常数** —— 它不设上界,只要求「合并这件事本身别把
|
|
234
|
+
* 脚手架抬高」。RSIH 的 ‖H‖≤B_H 需要一个数,是因为它必须允许学到上界、到顶再强制合并;
|
|
235
|
+
* 本仓已有 dedup 那一半(教训按 `## category: title` 幂等、规则按 id 幂等),
|
|
236
|
+
* 缺的只是另一半,而那一半不需要数:**闸只在「试图整合」那一刻开火**。
|
|
237
|
+
*
|
|
238
|
+
* 返回拒绝理由,合法时返回 null(同 validateBlastRadius 的签名)。
|
|
239
|
+
*/
|
|
240
|
+
export function validateMergeConvergence(proposal: {
|
|
241
|
+
filePath?: string
|
|
242
|
+
originalContent?: string
|
|
243
|
+
newContent?: string
|
|
244
|
+
merge?: boolean
|
|
245
|
+
}): string | null {
|
|
246
|
+
if (proposal.merge !== true) return null
|
|
247
|
+
// 无基线不是有增长:手工路径在文件不存在时正是这个形态(commands.ts 的宽松模式)。
|
|
248
|
+
if (!proposal.originalContent || !proposal.newContent) return null
|
|
249
|
+
|
|
250
|
+
const filePath = proposal.filePath ?? ''
|
|
251
|
+
const before = measureScaffold(filePath, proposal.originalContent)
|
|
252
|
+
const after = measureScaffold(filePath, proposal.newContent)
|
|
253
|
+
|
|
254
|
+
const rose: string[] = []
|
|
255
|
+
if (after.lessons > before.lessons) rose.push(`教训段数 ${before.lessons} → ${after.lessons}`)
|
|
256
|
+
if (after.rules > before.rules) rose.push(`规则条数 ${before.rules} → ${after.rules}`)
|
|
257
|
+
if (after.bytes > before.bytes) rose.push(`字节数 ${before.bytes} → ${after.bytes}`)
|
|
258
|
+
if (rose.length === 0) return null
|
|
259
|
+
|
|
260
|
+
return `合并型提案必须收敛,但脚手架增长了:${rose.join(';')}。`
|
|
261
|
+
}
|
|
262
|
+
|
|
198
263
|
// ── Sandbox ──
|
|
199
264
|
|
|
200
265
|
export class CrsiSandbox {
|
package/src/core/eval-harness.ts
CHANGED
|
@@ -21,14 +21,21 @@ import { PreFlightChecker } from './preflight-checker'
|
|
|
21
21
|
import { createDefaultPostFlightChecker } from './post-flight-checker'
|
|
22
22
|
import { WorkingMemory } from './working-memory'
|
|
23
23
|
import { RedTeam } from './red-team'
|
|
24
|
-
import {
|
|
24
|
+
import {
|
|
25
|
+
isProtectedPath,
|
|
26
|
+
validateBlastRadius,
|
|
27
|
+
validateMergeConvergence,
|
|
28
|
+
PROTECTED_CRITICAL_FILES,
|
|
29
|
+
} from './crsi-sandbox'
|
|
25
30
|
import {
|
|
26
31
|
produceRuleProposal,
|
|
27
32
|
MANAGED_RULES_FILE,
|
|
33
|
+
LESSONS_FILE,
|
|
28
34
|
buildLessonContent,
|
|
29
35
|
renderManagedRuleSource,
|
|
30
36
|
} from './crsi-producer'
|
|
31
37
|
import type { CrsiSignal } from './crsi-producer'
|
|
38
|
+
import { predictionHit } from './improvement-track'
|
|
32
39
|
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
33
40
|
|
|
34
41
|
// ── Types ──
|
|
@@ -43,6 +50,12 @@ export interface EvalResult {
|
|
|
43
50
|
detail?: string
|
|
44
51
|
/** 契约角色。缺省 neutral。 */
|
|
45
52
|
role?: ContractRole
|
|
53
|
+
/**
|
|
54
|
+
* anchor 契约的**定义处真源**。`role` 由此派生(见 runEval 末尾的回填)。
|
|
55
|
+
* `ANCHOR_CONTRACT_IDS` 退为**独立声明**,两者由 test/integrity/anchor-contract-wiring
|
|
56
|
+
* 守卫两向比对 —— 两处独立陈述同一件事,它们才可能不一致,守卫才有内容。
|
|
57
|
+
*/
|
|
58
|
+
anchor?: true
|
|
46
59
|
}
|
|
47
60
|
|
|
48
61
|
export interface EvalReport {
|
|
@@ -70,6 +83,8 @@ export const ANCHOR_CONTRACT_IDS: ReadonlySet<string> = new Set([
|
|
|
70
83
|
'red-team-zero-gaps',
|
|
71
84
|
'producer-rule-shape',
|
|
72
85
|
'producer-rule-idempotent',
|
|
86
|
+
'prediction-hit-truth-table',
|
|
87
|
+
'merge-convergence-gate',
|
|
73
88
|
'self-report-diagnostic',
|
|
74
89
|
])
|
|
75
90
|
|
|
@@ -82,10 +97,24 @@ export function regressedAnchors(results: EvalResult[]): string[] {
|
|
|
82
97
|
|
|
83
98
|
const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
|
|
84
99
|
|
|
100
|
+
/** 落盘的契约粒度投影 —— 只要 id/passed/role(EvalResult 的 description/detail 不落盘)。 */
|
|
101
|
+
export interface ContractResultRecord {
|
|
102
|
+
id: string
|
|
103
|
+
passed: boolean
|
|
104
|
+
role?: ContractRole
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** 一次评估的契约粒度快照:契约 id → 是否通过。 */
|
|
108
|
+
export type ContractSnapshot = Record<string, boolean>
|
|
109
|
+
|
|
110
|
+
function toContractResultRecord(r: EvalResult): ContractResultRecord {
|
|
111
|
+
return { id: r.id, passed: r.passed, ...(r.role ? { role: r.role } : {}) }
|
|
112
|
+
}
|
|
113
|
+
|
|
85
114
|
/** 追加一次评估分数到 rewards 日志(按奖励函数名键控)。 */
|
|
86
115
|
export function appendEvalScore(
|
|
87
116
|
name: string,
|
|
88
|
-
report: { score: number; passed: number; total: number },
|
|
117
|
+
report: { score: number; passed: number; total: number; results?: EvalResult[] },
|
|
89
118
|
): void {
|
|
90
119
|
try {
|
|
91
120
|
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
@@ -97,6 +126,9 @@ export function appendEvalScore(
|
|
|
97
126
|
score: report.score,
|
|
98
127
|
passed: report.passed,
|
|
99
128
|
total: report.total,
|
|
129
|
+
// 契约粒度(B1)。缺省不写该键:B1 之前落盘的旧记录没有它,
|
|
130
|
+
// 读取侧跳过 —— 这是向后兼容的承重判据,别改成 `results: []`。
|
|
131
|
+
...(report.results ? { results: report.results.map(toContractResultRecord) } : {}),
|
|
100
132
|
}) + '\n',
|
|
101
133
|
'utf-8',
|
|
102
134
|
)
|
|
@@ -120,6 +152,105 @@ export function getLastEvalScore(name: string): number | null {
|
|
|
120
152
|
}
|
|
121
153
|
}
|
|
122
154
|
|
|
155
|
+
/**
|
|
156
|
+
* 某奖励函数最近 n 次**按契约粒度**落盘的记录,新→旧。
|
|
157
|
+
*
|
|
158
|
+
* 只认带 `results` 的记录 —— B1 之前落盘的旧记录(只有聚合分数)被跳过,
|
|
159
|
+
* 于是调用方不必区分新旧形态。逐行容错:坏行跳过而不是让整条历史归零
|
|
160
|
+
* (`fixCache` 负责清理坏行)。
|
|
161
|
+
*/
|
|
162
|
+
export function getContractHistory(name: string, n = 3): ContractSnapshot[] {
|
|
163
|
+
try {
|
|
164
|
+
if (!existsSync(SCORES_FILE)) return []
|
|
165
|
+
const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
|
|
166
|
+
const out: ContractSnapshot[] = []
|
|
167
|
+
for (let i = lines.length - 1; i >= 0 && out.length < n; i--) {
|
|
168
|
+
let rec: { name?: string; results?: ContractResultRecord[] }
|
|
169
|
+
try {
|
|
170
|
+
rec = JSON.parse(lines[i]!) as { name?: string; results?: ContractResultRecord[] }
|
|
171
|
+
} catch {
|
|
172
|
+
continue
|
|
173
|
+
}
|
|
174
|
+
if (rec.name !== name || !Array.isArray(rec.results)) continue
|
|
175
|
+
const snap: ContractSnapshot = {}
|
|
176
|
+
for (const r of rec.results) {
|
|
177
|
+
if (typeof r?.id === 'string') snap[r.id] = r.passed === true
|
|
178
|
+
}
|
|
179
|
+
out.push(snap)
|
|
180
|
+
}
|
|
181
|
+
return out
|
|
182
|
+
} catch {
|
|
183
|
+
return []
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
export type ContractDelta = 'regressed' | 'fixed' | 'flaky' | 'new' | 'gone'
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* 纯函数:当前 run vs 历史 → 每条契约的变化。**只报变化**,未变化的契约不出现。
|
|
191
|
+
*
|
|
192
|
+
* delta 判据(`history` 新→旧):
|
|
193
|
+
* - 历史上没出现过 → `new`
|
|
194
|
+
* - 历史上 true/false 都出现过 → `flaky`(压过 regressed/fixed:抖动的契约
|
|
195
|
+
* 不该被报成「已修复」或「回归」)
|
|
196
|
+
* - 上次与本次相反 → `regressed`(上次 PASS→本次 FAIL)/ `fixed`
|
|
197
|
+
* - 本次没有但历史有 → `gone`
|
|
198
|
+
*
|
|
199
|
+
* 为什么 `flaky` 要压过相邻两次的比较:只比相邻两次会把一个每次都在翻的契约
|
|
200
|
+
* 误报成「真回归」,而那正是这个账本要区分开的东西。
|
|
201
|
+
*/
|
|
202
|
+
export function diffContractHistory(
|
|
203
|
+
current: ContractResultRecord[],
|
|
204
|
+
history: ContractSnapshot[],
|
|
205
|
+
): { id: string; delta: ContractDelta; role?: ContractRole }[] {
|
|
206
|
+
const out: { id: string; delta: ContractDelta; role?: ContractRole }[] = []
|
|
207
|
+
const seen = new Set<string>()
|
|
208
|
+
for (const c of current) {
|
|
209
|
+
seen.add(c.id)
|
|
210
|
+
const past = history.filter((h) => c.id in h).map((h) => h[c.id] === true)
|
|
211
|
+
let delta: ContractDelta
|
|
212
|
+
if (past.length === 0) delta = 'new'
|
|
213
|
+
else if (past.includes(true) && past.includes(false)) delta = 'flaky'
|
|
214
|
+
else if (past[0] === !c.passed) delta = c.passed ? 'fixed' : 'regressed'
|
|
215
|
+
else continue // 未变化
|
|
216
|
+
out.push({ id: c.id, delta, ...(c.role ? { role: c.role } : {}) })
|
|
217
|
+
}
|
|
218
|
+
for (const h of history) {
|
|
219
|
+
for (const id of Object.keys(h)) {
|
|
220
|
+
if (seen.has(id)) continue
|
|
221
|
+
seen.add(id)
|
|
222
|
+
out.push({ id, delta: 'gone' })
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
return out
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* 把 diffContractHistory 的结果渲染成展示行。纯函数——不做 I/O、不读时钟。
|
|
230
|
+
* 返回空数组表示「无变化」,调用方据此决定是否打印标题。
|
|
231
|
+
*/
|
|
232
|
+
export function renderContractDiff(
|
|
233
|
+
deltas: { id: string; delta: ContractDelta; role?: ContractRole }[],
|
|
234
|
+
): string[] {
|
|
235
|
+
const text: Record<ContractDelta, string> = {
|
|
236
|
+
regressed: '上次 PASS,本次 FAIL(回归)',
|
|
237
|
+
fixed: '上次 FAIL,本次 PASS(已修复)',
|
|
238
|
+
flaky: '近几次结果不一致(抖动)',
|
|
239
|
+
new: '本次新增的契约',
|
|
240
|
+
gone: '本次未出现(已移出契约集)',
|
|
241
|
+
}
|
|
242
|
+
const icon: Record<ContractDelta, string> = {
|
|
243
|
+
regressed: '❌',
|
|
244
|
+
fixed: '✅',
|
|
245
|
+
flaky: '⚠️',
|
|
246
|
+
new: '🆕',
|
|
247
|
+
gone: '➖',
|
|
248
|
+
}
|
|
249
|
+
return deltas.map(
|
|
250
|
+
(d) => `${icon[d.delta]} ${d.id} ← ${text[d.delta]}${d.role ? ` \`${d.role}\`` : ''}`,
|
|
251
|
+
)
|
|
252
|
+
}
|
|
253
|
+
|
|
123
254
|
// ── Harness ──
|
|
124
255
|
|
|
125
256
|
/** 构建隔离组件,避免读用户 ~/.mipham 运行时状态。 */
|
|
@@ -147,6 +278,7 @@ export function runEval(): EvalReport {
|
|
|
147
278
|
id: 'rule-timeout',
|
|
148
279
|
description: '内置 timeout 规则命中低超时的 npm install',
|
|
149
280
|
passed: timeout.modified.timeout === 300000,
|
|
281
|
+
anchor: true,
|
|
150
282
|
})
|
|
151
283
|
|
|
152
284
|
const gitForce = ruleEngine.intercept('Bash', {
|
|
@@ -157,6 +289,7 @@ export function runEval(): EvalReport {
|
|
|
157
289
|
id: 'rule-git-force',
|
|
158
290
|
description: 'git --force 触发告警',
|
|
159
291
|
passed: gitForce.warnings.length > 0,
|
|
292
|
+
anchor: true,
|
|
160
293
|
})
|
|
161
294
|
|
|
162
295
|
const disabledRule: import('./rule-engine').ToolRule = {
|
|
@@ -174,6 +307,7 @@ export function runEval(): EvalReport {
|
|
|
174
307
|
id: 'rule-disabled-skip',
|
|
175
308
|
description: '禁用规则被跳过',
|
|
176
309
|
passed: disabled.warnings.length === 0,
|
|
310
|
+
anchor: true,
|
|
177
311
|
})
|
|
178
312
|
|
|
179
313
|
// ── 宪法(ground truth:8 原则 + facet 映射 + 愿力序言) ──
|
|
@@ -182,6 +316,7 @@ export function runEval(): EvalReport {
|
|
|
182
316
|
id: 'constitution-8-principles',
|
|
183
317
|
description: '宪法含 8 条原则',
|
|
184
318
|
passed: principles.length === 8,
|
|
319
|
+
anchor: true,
|
|
185
320
|
})
|
|
186
321
|
|
|
187
322
|
const prajna = principles.filter((p) => p.facet === 'prajna').length
|
|
@@ -191,12 +326,14 @@ export function runEval(): EvalReport {
|
|
|
191
326
|
id: 'constitution-facets',
|
|
192
327
|
description: 'facet 映射 智3 / 金刚5 / 悲0',
|
|
193
328
|
passed: prajna === 3 && vajra === 5 && karuna === 0,
|
|
329
|
+
anchor: true,
|
|
194
330
|
})
|
|
195
331
|
|
|
196
332
|
results.push({
|
|
197
333
|
id: 'constitution-preamble',
|
|
198
334
|
description: '愿力序言已注入',
|
|
199
335
|
passed: !!DEFAULT_CONSTITUTION.preamble && DEFAULT_CONSTITUTION.preamble.includes('悲'),
|
|
336
|
+
anchor: true,
|
|
200
337
|
})
|
|
201
338
|
|
|
202
339
|
// ── 沙箱只读边界(ground truth:受保护路径被拒) ──
|
|
@@ -206,7 +343,12 @@ export function runEval(): EvalReport {
|
|
|
206
343
|
['sandbox-protected-machinery', 'apps/cli/src/core/crsi-sandbox.ts'],
|
|
207
344
|
]
|
|
208
345
|
for (const [id, path] of protectedChecks) {
|
|
209
|
-
results.push({
|
|
346
|
+
results.push({
|
|
347
|
+
id,
|
|
348
|
+
description: `受保护路径被拒: ${path}`,
|
|
349
|
+
passed: isProtectedPath(path),
|
|
350
|
+
anchor: true,
|
|
351
|
+
})
|
|
210
352
|
}
|
|
211
353
|
|
|
212
354
|
// ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
|
|
@@ -216,6 +358,7 @@ export function runEval(): EvalReport {
|
|
|
216
358
|
description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
|
|
217
359
|
passed: unprotected.length === 0,
|
|
218
360
|
...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
|
|
361
|
+
anchor: true,
|
|
219
362
|
})
|
|
220
363
|
|
|
221
364
|
// ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
|
|
@@ -226,6 +369,7 @@ export function runEval(): EvalReport {
|
|
|
226
369
|
validateBlastRadius({ blastRadius: undefined }) !== null &&
|
|
227
370
|
validateBlastRadius({ blastRadius: [] }) !== null &&
|
|
228
371
|
validateBlastRadius({ blastRadius: ['apps/cli/src/foo.ts'] }) === null,
|
|
372
|
+
anchor: true,
|
|
229
373
|
})
|
|
230
374
|
|
|
231
375
|
// ── 安全(ground truth:16 攻击零漏过) ──
|
|
@@ -235,6 +379,7 @@ export function runEval(): EvalReport {
|
|
|
235
379
|
description: '16 个对抗场景零漏过',
|
|
236
380
|
passed: redTeam.passedThrough === 0,
|
|
237
381
|
detail: `score=${redTeam.score}, passedThrough=${redTeam.passedThrough}, falsePositives=${redTeam.falsePositives}`,
|
|
382
|
+
anchor: true,
|
|
238
383
|
})
|
|
239
384
|
|
|
240
385
|
// ── producer 行为(ground truth:固化规则产出正确 shape + 幂等) ──
|
|
@@ -255,6 +400,7 @@ export function runEval(): EvalReport {
|
|
|
255
400
|
ruleProposal.newContent.includes("source: 'managed'") &&
|
|
256
401
|
ruleProposal.newContent.includes('timeout: 300000') &&
|
|
257
402
|
ruleProposal.newContent.includes('enabled: true'),
|
|
403
|
+
anchor: true,
|
|
258
404
|
})
|
|
259
405
|
|
|
260
406
|
results.push({
|
|
@@ -262,6 +408,7 @@ export function runEval(): EvalReport {
|
|
|
262
408
|
description: '同名规则重复产出被拒(幂等)',
|
|
263
409
|
passed:
|
|
264
410
|
ruleProposal !== null && produceRuleProposal(frozenSignal, ruleProposal.newContent) === null,
|
|
411
|
+
anchor: true,
|
|
265
412
|
})
|
|
266
413
|
|
|
267
414
|
// ── 组件归因(ground truth:缺省 experiential、显式组件透传、非 experiential 不进 managed-rule) ──
|
|
@@ -334,6 +481,46 @@ export function runEval(): EvalReport {
|
|
|
334
481
|
results.push({ ...judgeBehaviorTask(task, ruleEngine), role: 'target' })
|
|
335
482
|
}
|
|
336
483
|
|
|
484
|
+
// ── ε 预测命中真值表(ground truth:命中判据不叠加统计阈值) ──
|
|
485
|
+
// `(20, 20)` 那条**承重**:判据是 `deltaMean >= predicted` 而 `>=` 与 `>` 只在
|
|
486
|
+
// `predicted === deltaMean` 处分歧 ⇒ 少了它,「把 >= 翻成 >」在契约上不可观测。
|
|
487
|
+
results.push({
|
|
488
|
+
id: 'prediction-hit-truth-table',
|
|
489
|
+
description: 'predictionHit 真值表(返回值):未达不算、达到或恰好相等算命中、缺席恒 false',
|
|
490
|
+
passed:
|
|
491
|
+
predictionHit(50, 20) === false &&
|
|
492
|
+
predictionHit(10, 20) === true &&
|
|
493
|
+
predictionHit(20, 20) === true &&
|
|
494
|
+
predictionHit(undefined, 20) === false,
|
|
495
|
+
anchor: true,
|
|
496
|
+
})
|
|
497
|
+
|
|
498
|
+
// ── B_H 合并型收敛闸(ground truth:净增被拒、删二增一通过、非合并型不受此闸) ──
|
|
499
|
+
results.push({
|
|
500
|
+
id: 'merge-convergence-gate',
|
|
501
|
+
description: '合并型净增被拒、删二增一通过、merge=false 净增通过',
|
|
502
|
+
passed:
|
|
503
|
+
validateMergeConvergence({
|
|
504
|
+
filePath: LESSONS_FILE,
|
|
505
|
+
originalContent: '## a: 1\n\n## b: 2\n',
|
|
506
|
+
newContent: '## a: 1\n\n## b: 2\n\n## c: 3\n',
|
|
507
|
+
merge: true,
|
|
508
|
+
}) !== null &&
|
|
509
|
+
validateMergeConvergence({
|
|
510
|
+
filePath: LESSONS_FILE,
|
|
511
|
+
originalContent: '## a: 1\n\n## b: 2\n',
|
|
512
|
+
newContent: '## ab: merged\n',
|
|
513
|
+
merge: true,
|
|
514
|
+
}) === null &&
|
|
515
|
+
validateMergeConvergence({
|
|
516
|
+
filePath: LESSONS_FILE,
|
|
517
|
+
originalContent: '## a: 1\n',
|
|
518
|
+
newContent: '## a: 1\n\n## b: 2\n',
|
|
519
|
+
merge: false,
|
|
520
|
+
}) === null,
|
|
521
|
+
anchor: true,
|
|
522
|
+
})
|
|
523
|
+
|
|
337
524
|
// ── 自报分数只作诊断:评分路径无 LLM,分数来自 ground-truth 契约而非模型自报 ──
|
|
338
525
|
// anchor 锁死「评分组件不暴露 LLM 的 chat 能力」。4 个组件(ruleEngine/constitution/
|
|
339
526
|
// errorDB/preflight)都是确定性组件(runEval 同步评分)。若未来有人把 LLM 注入评分
|
|
@@ -347,11 +534,13 @@ export function runEval(): EvalReport {
|
|
|
347
534
|
id: 'self-report-diagnostic',
|
|
348
535
|
description: '评分无 LLM:机制哨兵组件不暴露 chat 能力(分数只来自 ground-truth,非模型自报)',
|
|
349
536
|
passed: !llmInjected,
|
|
537
|
+
anchor: true,
|
|
350
538
|
})
|
|
351
539
|
|
|
352
|
-
// 角色标注:anchor
|
|
540
|
+
// 角色标注:anchor 由契约**定义处内联的标记**派生(真源),
|
|
541
|
+
// ANCHOR_CONTRACT_IDS 退为独立声明 —— 两者由 anchor-contract-wiring 守卫两向比对。
|
|
353
542
|
for (const r of results) {
|
|
354
|
-
if (
|
|
543
|
+
if (r.anchor) r.role = 'anchor'
|
|
355
544
|
}
|
|
356
545
|
|
|
357
546
|
// anchor 自检(ground truth:所有 anchor 契约必须全绿,否则门拒)。
|
|
@@ -22,6 +22,17 @@ export interface ImprovementReport {
|
|
|
22
22
|
noise: number
|
|
23
23
|
minEffect: number
|
|
24
24
|
verdict: ImprovementVerdict
|
|
25
|
+
/** ε:提交者事前写下的预期提升点数。缺席 = 该记录没有预登记。 */
|
|
26
|
+
predictedDelta?: number
|
|
27
|
+
/** 预测是否命中。与 predictedDelta 同时出现、同时缺席(JSON 序列化会丢掉 undefined 键)。 */
|
|
28
|
+
predictionHit?: boolean
|
|
29
|
+
/**
|
|
30
|
+
* B2 代价维:与分数数组逐项对齐的前/后耗时。**只记录,不进任何门禁** ——
|
|
31
|
+
* `verdict` / `deltaMean` / `minEffect` 一律不看这两个字段。
|
|
32
|
+
* 缺席(而非空数组)= 该记录早于代价维落地,或该样本未记代价。
|
|
33
|
+
*/
|
|
34
|
+
baselineDurations?: number[]
|
|
35
|
+
postDurations?: number[]
|
|
25
36
|
}
|
|
26
37
|
|
|
27
38
|
export interface ImprovementRecord extends ImprovementReport {
|
|
@@ -55,6 +66,7 @@ function stdDev(xs: number[]): number {
|
|
|
55
66
|
export function buildImprovementReport(
|
|
56
67
|
sample: SkillDeltaSample,
|
|
57
68
|
changeSet: string[],
|
|
69
|
+
predicted?: number,
|
|
58
70
|
): ImprovementReport {
|
|
59
71
|
const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
|
|
60
72
|
const noise = stdDev(sample.baselineScores)
|
|
@@ -70,6 +82,19 @@ export function buildImprovementReport(
|
|
|
70
82
|
noise,
|
|
71
83
|
minEffect,
|
|
72
84
|
verdict,
|
|
85
|
+
// 两个字段同生同灭:缺席预测必须**键不存在**,而不是 predictionHit: false。
|
|
86
|
+
// 理由**不是**「predictionHit: false 会被算进分母」—— 分母只认 `predictedDelta !== undefined`
|
|
87
|
+
//(只写 `predictionHit: false` 而不写 `predictedDelta` 会被整条忽略)。真正的风险是**反过来的半条**:
|
|
88
|
+
// 有 `predictedDelta` 而无 `predictionHit` ⇒ 该条进了分母,却永远不可能被算成命中
|
|
89
|
+
//(分子只认 `predictionHit === true`),等于一条静默的「未命中」。
|
|
90
|
+
...(predicted !== undefined
|
|
91
|
+
? { predictedDelta: predicted, predictionHit: predictionHit(predicted, deltaMean) }
|
|
92
|
+
: {}),
|
|
93
|
+
// 代价维:**两条同生同灭**,与上面 ε 那条同理 —— 只写一半(有 baselineDurations
|
|
94
|
+
// 而无 postDurations)会让「代价」这件事在记录里既非有也非无,读侧无从判断。
|
|
95
|
+
...(sample.baselineDurations !== undefined && sample.postDurations !== undefined
|
|
96
|
+
? { baselineDurations: sample.baselineDurations, postDurations: sample.postDurations }
|
|
97
|
+
: {}),
|
|
73
98
|
}
|
|
74
99
|
}
|
|
75
100
|
|
|
@@ -107,6 +132,52 @@ export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
|
|
|
107
132
|
return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
|
|
108
133
|
}
|
|
109
134
|
|
|
135
|
+
/**
|
|
136
|
+
* 预测命中:事前写下的点数被实际达到。缺席预测(undefined)不计入。
|
|
137
|
+
*
|
|
138
|
+
* 刻意**不叠加 `minEffect`** —— ε 是提交者自己写下的数,判据就是「达到没达到」;
|
|
139
|
+
* 再套一层统计阈值会让两个数打架,且使「命中」不可复算。
|
|
140
|
+
*/
|
|
141
|
+
export function predictionHit(predicted: number | undefined, deltaMean: number): boolean {
|
|
142
|
+
return predicted !== undefined && deltaMean >= predicted
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* ε 命中率。分母 = **有预测的记录数**(判据取 `predictedDelta !== undefined`),
|
|
147
|
+
* 复用既有 wilsonInterval。无预测的记录既不入分子也不入分母。
|
|
148
|
+
*/
|
|
149
|
+
export function predictionHitRate(records: ImprovementRecord[]): {
|
|
150
|
+
total: number
|
|
151
|
+
hits: number
|
|
152
|
+
rate: number
|
|
153
|
+
lo: number
|
|
154
|
+
hi: number
|
|
155
|
+
} {
|
|
156
|
+
const judged = records.filter((r) => r.predictedDelta !== undefined)
|
|
157
|
+
const total = judged.length
|
|
158
|
+
const hits = judged.filter((r) => r.predictionHit === true).length
|
|
159
|
+
const { lo, hi } = wilsonInterval(hits, total)
|
|
160
|
+
return { total, hits, rate: total === 0 ? 0 : hits / total, lo, hi }
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* 代价维的只读展示(B2):前/后均值一行。两个耗时数组缺席 → null,调用方据此整行不打印。
|
|
165
|
+
*
|
|
166
|
+
* **刻意不下结论**:倍数只是描述,不是判据。把「代价过高」变成 verdict 的一部分需要
|
|
167
|
+
* 样本量支撑,而当前 k 默认 3、`minEffect` 已经要 `max(20, 2×噪声)` —— 再塞一个维度
|
|
168
|
+
* 只会把统计问题变得更糟。所以这里只回答「花了多少」,不回答「值不值」。
|
|
169
|
+
*
|
|
170
|
+
* 基线均值为 0 时**不给倍数**:那个比值是 ∞(或 0/0 的 NaN),打出来是假读数。
|
|
171
|
+
*/
|
|
172
|
+
export function formatCostLine(report: ImprovementReport): string | null {
|
|
173
|
+
const { baselineDurations: before_, postDurations: after_ } = report
|
|
174
|
+
if (before_ === undefined || after_ === undefined) return null
|
|
175
|
+
const before = Math.round(mean(before_))
|
|
176
|
+
const after = Math.round(mean(after_))
|
|
177
|
+
const line = `⏱️ 代价: 均值 ${before}ms → ${after}ms`
|
|
178
|
+
return before > 0 ? `${line}(×${(after / before).toFixed(1)})` : line
|
|
179
|
+
}
|
|
180
|
+
|
|
110
181
|
// ── 台账 ──
|
|
111
182
|
|
|
112
183
|
export function improvementPath(): string {
|
|
@@ -68,6 +68,8 @@ export interface TaskPerformanceReport {
|
|
|
68
68
|
score: number
|
|
69
69
|
results: TaskPerformanceResult[]
|
|
70
70
|
failures: string[]
|
|
71
|
+
/** B2 代价维:整轮(LLM 生成 + 冻结测试判定)的墙钟耗时。只记录,不参与任何判定。 */
|
|
72
|
+
durationMs: number
|
|
71
73
|
}
|
|
72
74
|
|
|
73
75
|
/** 剥掉 LLM 可能包裹的 markdown 代码块(```ts ... ```),拿到裸代码。 */
|
|
@@ -101,6 +103,7 @@ export async function runTaskPerformance(
|
|
|
101
103
|
const wanted = opts?.skill?.name
|
|
102
104
|
const tasks = loadPerformanceTasks().filter((t) => (t.skill ?? undefined) === wanted)
|
|
103
105
|
const results: TaskPerformanceResult[] = []
|
|
106
|
+
const startedAt = Date.now()
|
|
104
107
|
for (const task of tasks) {
|
|
105
108
|
const code = await collectGeneratedCode(llm, task.prompt, opts?.skill?.text)
|
|
106
109
|
if (!code) {
|
|
@@ -127,6 +130,7 @@ export async function runTaskPerformance(
|
|
|
127
130
|
score: results.length > 0 ? Math.round((passed / results.length) * 100) : 100,
|
|
128
131
|
results,
|
|
129
132
|
failures: results.filter((r) => !r.passed).map((r) => r.id),
|
|
133
|
+
durationMs: Date.now() - startedAt,
|
|
130
134
|
}
|
|
131
135
|
}
|
|
132
136
|
|
|
@@ -210,6 +214,13 @@ export interface SkillDeltaSample {
|
|
|
210
214
|
skillName: string
|
|
211
215
|
baselineScores: number[]
|
|
212
216
|
postScores: number[]
|
|
217
|
+
/**
|
|
218
|
+
* B2 代价维:与分数数组**逐项对齐**的耗时(第 i 项就是产出第 i 个分数的那次采样)。
|
|
219
|
+
* 缺席 = 该样本未记代价(旧记录 / 旧调用点),读侧须按「缺席」处理、不得当成 0。
|
|
220
|
+
* 只记录,不参与改进判定 —— 样本量 k=3 时再加一个维度只会让统计问题更糟。
|
|
221
|
+
*/
|
|
222
|
+
baselineDurations?: number[]
|
|
223
|
+
postDurations?: number[]
|
|
213
224
|
}
|
|
214
225
|
|
|
215
226
|
/**
|
|
@@ -227,18 +238,28 @@ export async function measureSkillDeltaRepeated(
|
|
|
227
238
|
const k = opts?.k ?? 3
|
|
228
239
|
const baselineScores: number[] = []
|
|
229
240
|
const postScores: number[] = []
|
|
241
|
+
const baselineDurations: number[] = []
|
|
242
|
+
const postDurations: number[] = []
|
|
230
243
|
for (let i = 0; i < k; i++) {
|
|
231
244
|
const r = await runTaskPerformance(llm, {
|
|
232
245
|
skill: { name: resolved.skillName, text: resolved.baselineText },
|
|
233
246
|
})
|
|
234
247
|
baselineScores.push(r.score)
|
|
248
|
+
baselineDurations.push(r.durationMs)
|
|
235
249
|
}
|
|
236
250
|
for (let i = 0; i < k; i++) {
|
|
237
251
|
const r = await runTaskPerformance(llm, {
|
|
238
252
|
skill: { name: resolved.skillName, text: resolved.postText },
|
|
239
253
|
})
|
|
240
254
|
postScores.push(r.score)
|
|
255
|
+
postDurations.push(r.durationMs)
|
|
241
256
|
}
|
|
242
257
|
|
|
243
|
-
return {
|
|
258
|
+
return {
|
|
259
|
+
skillName: resolved.skillName,
|
|
260
|
+
baselineScores,
|
|
261
|
+
postScores,
|
|
262
|
+
baselineDurations,
|
|
263
|
+
postDurations,
|
|
264
|
+
}
|
|
244
265
|
}
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
export const PACKAGE_NAME = '@miphamai/cli' as const
|
|
10
10
|
|
|
11
11
|
/** 当前发布版本 */
|
|
12
|
-
export const PACKAGE_VERSION = '0.
|
|
12
|
+
export const PACKAGE_VERSION = '0.83.0' as const
|
|
13
13
|
|
|
14
14
|
/** npm install 全局安装命令 */
|
|
15
15
|
export const NPM_INSTALL_COMMAND = `npm install -g ${PACKAGE_NAME}` as const
|
package/src/ui/commands.ts
CHANGED
|
@@ -37,7 +37,13 @@ import {
|
|
|
37
37
|
MANAGED_RULES_FILE,
|
|
38
38
|
} from '../core/crsi-producer'
|
|
39
39
|
import { prefilterProposal } from '../core/proposal-guard'
|
|
40
|
-
import {
|
|
40
|
+
import {
|
|
41
|
+
runEval,
|
|
42
|
+
appendEvalScore,
|
|
43
|
+
getContractHistory,
|
|
44
|
+
diffContractHistory,
|
|
45
|
+
renderContractDiff,
|
|
46
|
+
} from '../core/eval-harness'
|
|
41
47
|
import { listRewardFns } from '../core/reward-fn'
|
|
42
48
|
import {
|
|
43
49
|
runTaskPerformance,
|
|
@@ -47,9 +53,11 @@ import {
|
|
|
47
53
|
import { randomUUID } from 'node:crypto'
|
|
48
54
|
import {
|
|
49
55
|
buildImprovementReport,
|
|
56
|
+
formatCostLine,
|
|
50
57
|
appendImprovement,
|
|
51
58
|
readImprovements,
|
|
52
59
|
improvementRate,
|
|
60
|
+
predictionHitRate,
|
|
53
61
|
setPendingVerdict,
|
|
54
62
|
getPendingVerdict,
|
|
55
63
|
shouldBlockApproval,
|
|
@@ -804,6 +812,25 @@ const crsiStatsCmd: CommandHandler = async (ctx) => {
|
|
|
804
812
|
}
|
|
805
813
|
}
|
|
806
814
|
|
|
815
|
+
// ── ε 预测命中(prose 路径) ──
|
|
816
|
+
// 作废条款:样本不足就不下结论;台账攒够 20 条而判定样本仍 < 5 ⇒ 明写机制失效。
|
|
817
|
+
// 只打印、不写台账:本命令是只读的,而该结论每次都能从同一份 improvements.jsonl 重算出来。
|
|
818
|
+
const records = readImprovements()
|
|
819
|
+
const pred = predictionHitRate(records)
|
|
820
|
+
lines.push('')
|
|
821
|
+
lines.push('### ε 预测命中(prose 路径)')
|
|
822
|
+
if (pred.total < 5) {
|
|
823
|
+
lines.push(`样本不足(判定记录 ${pred.total} 条,需 ≥ 5)—— 不下结论。`)
|
|
824
|
+
if (records.length >= 20) {
|
|
825
|
+
lines.push('⚠️ ε 机制失效:prose 路径使用率过低(记录总数已达 20 而判定样本仍 < 5)。')
|
|
826
|
+
}
|
|
827
|
+
} else {
|
|
828
|
+
lines.push(
|
|
829
|
+
`命中率: ${pred.hits}/${pred.total} (${(pred.rate * 100).toFixed(0)}%, ` +
|
|
830
|
+
`Wilson 95% [${(pred.lo * 100).toFixed(0)}%, ${(pred.hi * 100).toFixed(0)}%])`,
|
|
831
|
+
)
|
|
832
|
+
}
|
|
833
|
+
|
|
807
834
|
return { content: lines.join('\n') }
|
|
808
835
|
}
|
|
809
836
|
|
|
@@ -881,9 +908,12 @@ const crsiModifyCmd: CommandHandler = async (ctx, args) => {
|
|
|
881
908
|
? 'regressed ⚠️'
|
|
882
909
|
: 'inconclusive'
|
|
883
910
|
const sign = report.deltaMean >= 0 ? '+' : ''
|
|
911
|
+
// B2 代价维:只展示、不进判定(`formatCostLine` 缺席时返回 null ⇒ 整行不打)。
|
|
912
|
+
const costLine = formatCostLine(report)
|
|
884
913
|
improvementLine =
|
|
885
914
|
`\n📊 改进判定: ${label} (delta ${sign}${report.deltaMean.toFixed(1)}, 噪声 ${report.noise.toFixed(1)}, 阈值 ${report.minEffect.toFixed(1)})` +
|
|
886
915
|
`\n 改进率: ${rate.improved}/${rate.total} (${(rate.rate * 100).toFixed(0)}%, Wilson 95% [${(rate.lo * 100).toFixed(0)}%, ${(rate.hi * 100).toFixed(0)}%])` +
|
|
916
|
+
(costLine ? `\n${costLine}` : '') +
|
|
887
917
|
(report.verdict === 'regressed' ? '\n ⚠️ 任务表现倒退:--approve 将被拒绝。' : '')
|
|
888
918
|
}
|
|
889
919
|
} catch {
|
|
@@ -970,11 +1000,40 @@ const crsiProposeCmd: CommandHandler = async (ctx, args) => {
|
|
|
970
1000
|
return { content: `❌ 生成失败(phase: ${result.phase})。\n${result.error ?? ''}` }
|
|
971
1001
|
}
|
|
972
1002
|
|
|
1003
|
+
// ε 预登记落地:prose 路径此前**不测量**(measureSkillDeltaRepeated 全仓库只有手工路径一个
|
|
1004
|
+
// 调用点)⇒ ε 曾在 A 流程登记、判定侧在 B 流程,两端永不相遇。这里补上测量。
|
|
1005
|
+
// 成本:每次提案多 6 次 LLM 调用(同手工路径 :867 的注释)。
|
|
1006
|
+
let predictionLine = ''
|
|
1007
|
+
try {
|
|
1008
|
+
const sample = await measureSkillDeltaRepeated(llm, {
|
|
1009
|
+
filePath: proposal.filePath,
|
|
1010
|
+
originalContent: proposal.originalContent,
|
|
1011
|
+
newContent: proposal.newContent,
|
|
1012
|
+
})
|
|
1013
|
+
if (sample) {
|
|
1014
|
+
const report = buildImprovementReport(sample, [proposal.filePath], proposal.expectedEffect)
|
|
1015
|
+
setPendingVerdict(report.verdict)
|
|
1016
|
+
appendImprovement({ ...report, id: randomUUID(), timestamp: new Date().toISOString() })
|
|
1017
|
+
if (report.predictionHit !== undefined) {
|
|
1018
|
+
predictionLine =
|
|
1019
|
+
`\n🎯 ε 预测命中: ${report.predictionHit ? '命中 ✅' : '未命中 ⚠️'}` +
|
|
1020
|
+
`(预测 ${report.predictedDelta},实际 delta ${report.deltaMean.toFixed(1)})`
|
|
1021
|
+
}
|
|
1022
|
+
// B2 代价维:**与手工路径同一行读数**,两条渲染路径都接(只接一条即本仓库记过的
|
|
1023
|
+
// 「局部正确全局遗漏」)。ε 行可以有、代价行可以没有,故各自独立追加。
|
|
1024
|
+
const costLine = formatCostLine(report)
|
|
1025
|
+
if (costLine) predictionLine += `\n${costLine}`
|
|
1026
|
+
}
|
|
1027
|
+
} catch {
|
|
1028
|
+
// 测量失败(LLM 不可用等)不阻断提案流程 —— 与手工路径 :889 的处置一致。
|
|
1029
|
+
}
|
|
1030
|
+
|
|
973
1031
|
appendProseProposal({ id, filePath: proposal.filePath, timestamp: new Date().toISOString() })
|
|
974
1032
|
|
|
975
1033
|
return {
|
|
976
1034
|
content:
|
|
977
1035
|
`✅ 已生成散文提议并跑过测试。审阅 diff:\n\n${result.diff}\n\n` +
|
|
1036
|
+
predictionLine +
|
|
978
1037
|
'/crsi modify --approve 合并 | /crsi modify --reject 丢弃',
|
|
979
1038
|
}
|
|
980
1039
|
}
|
|
@@ -1087,14 +1146,26 @@ const crsiEvalCmd: CommandHandler = async (ctx, args) => {
|
|
|
1087
1146
|
content: `❌ 未知 reward: ${rewardName}。可用: ${fns.map((f) => f.name).join(', ')}`,
|
|
1088
1147
|
}
|
|
1089
1148
|
}
|
|
1149
|
+
// 先读后写:比较基准是「上一次已落盘的态」,不依赖刚写进去那条排第几。
|
|
1150
|
+
const prev = getContractHistory(fn.name)
|
|
1090
1151
|
const report = await fn.evaluate()
|
|
1152
|
+
const deltaLines = report.results
|
|
1153
|
+
? renderContractDiff(diffContractHistory(report.results, prev))
|
|
1154
|
+
: []
|
|
1091
1155
|
appendEvalScore(fn.name, report)
|
|
1092
1156
|
return {
|
|
1093
|
-
content:
|
|
1157
|
+
content: [
|
|
1158
|
+
`得分 **${report.score}/100** (${report.passed}/${report.total})`,
|
|
1159
|
+
`失败: ${report.failures.join(', ') || '无'}`,
|
|
1160
|
+
...(deltaLines.length > 0 ? ['', '### 与上次相比', ...deltaLines] : []),
|
|
1161
|
+
].join('\n'),
|
|
1094
1162
|
}
|
|
1095
1163
|
}
|
|
1096
1164
|
|
|
1165
|
+
// 先读后写(同上):基准是上一次落盘的态。
|
|
1166
|
+
const prevSnapshot = getContractHistory('mechanism-sentinel')
|
|
1097
1167
|
const report = runEval()
|
|
1168
|
+
const deltaLines = renderContractDiff(diffContractHistory(report.results, prevSnapshot))
|
|
1098
1169
|
appendEvalScore('mechanism-sentinel', report)
|
|
1099
1170
|
|
|
1100
1171
|
const lines: string[] = ['## 🧪 CRSI Eval Harness', '']
|
|
@@ -1109,6 +1180,11 @@ const crsiEvalCmd: CommandHandler = async (ctx, args) => {
|
|
|
1109
1180
|
if (report.failures.length > 0) {
|
|
1110
1181
|
lines.push('', `❌ 失败任务: ${report.failures.join(', ')}`)
|
|
1111
1182
|
}
|
|
1183
|
+
// 只读展示:账本按契约粒度落盘后,才能回答「是哪条契约翻的」。
|
|
1184
|
+
// 这里不做任何自动决策——闸门仍只看 regressedAnchors / score。
|
|
1185
|
+
if (deltaLines.length > 0) {
|
|
1186
|
+
lines.push('', '### 与上次相比', ...deltaLines)
|
|
1187
|
+
}
|
|
1112
1188
|
|
|
1113
1189
|
// 奖励函数注册表(reward function = policy→feedback 抽象可见)
|
|
1114
1190
|
// 传 llm 列出完整注册表(task-performance 需 llm 才能跑,但构造它零 LLM 调用)。
|