@miphamai/cli 0.81.9 → 0.82.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/crsi-modify.ts +34 -1
- package/src/core/crsi-producer.ts +77 -7
- package/src/core/crsi-sandbox.ts +65 -0
- package/src/core/eval-harness.ts +77 -4
- package/src/core/improvement-track.ts +41 -0
- package/src/shared/package-info.ts +1 -1
- package/src/ui/commands.ts +45 -0
package/package.json
CHANGED
package/src/core/crsi-modify.ts
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
|
-
import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
|
|
16
|
+
import { CrsiSandbox, validateBlastRadius, validateMergeConvergence } from './crsi-sandbox'
|
|
17
17
|
import type { CrsiModificationResult } from './crsi-sandbox'
|
|
18
18
|
import { appendEvalScore, getLastEvalScore, regressedAnchors } from './eval-harness'
|
|
19
19
|
import { mechanismSentinel, type RewardFn } from './reward-fn'
|
|
@@ -36,6 +36,19 @@ export interface CrsiProposal {
|
|
|
36
36
|
* 局部正确、全局遗漏。自修改前必须摸清并声明全部受影响路径,否则 fail-closed 拒绝。
|
|
37
37
|
*/
|
|
38
38
|
blastRadius?: string[]
|
|
39
|
+
/**
|
|
40
|
+
* ε:提交者**事前**写下的预期效果(任务表现提升点数)。
|
|
41
|
+
* 缺席 = 不预测。判定见 improvement-track 的 predictionHit。
|
|
42
|
+
*/
|
|
43
|
+
expectedEffect?: number
|
|
44
|
+
/** R:风险声明(这次改动可能在哪方面变差)。缺席 = 未声明。 */
|
|
45
|
+
risk?: string
|
|
46
|
+
/**
|
|
47
|
+
* 声明这是一次**合并型**提案(整合已有内容,而非新增)。
|
|
48
|
+
* 只有它为 true 时 `validateMergeConvergence` 才开火 —— 学习本身就是增长,
|
|
49
|
+
* 对新增型设非增长约束等于永久禁掉 `/crsi propose`。
|
|
50
|
+
*/
|
|
51
|
+
merge?: boolean
|
|
39
52
|
}
|
|
40
53
|
|
|
41
54
|
// ── Pending proposal registry (两阶段闸门) ──
|
|
@@ -70,6 +83,26 @@ export async function runCrsiModification(
|
|
|
70
83
|
}
|
|
71
84
|
}
|
|
72
85
|
|
|
86
|
+
// B_H 收敛闸:合并型提案不得抬高脚手架成本。
|
|
87
|
+
// 位置与 blast radius 闸同序 —— 都在 worktree 之前,纯字符串比较、零磁盘 I/O、零副作用。
|
|
88
|
+
const convergenceError = validateMergeConvergence(proposal)
|
|
89
|
+
if (convergenceError) {
|
|
90
|
+
return {
|
|
91
|
+
modification: {
|
|
92
|
+
id: 'crsi-mod-rejected-merge-convergence',
|
|
93
|
+
description: proposal.description,
|
|
94
|
+
filePath: proposal.filePath,
|
|
95
|
+
newContent: proposal.newContent,
|
|
96
|
+
originalContent: proposal.originalContent ?? '',
|
|
97
|
+
crsiInsightId: proposal.crsiInsightId,
|
|
98
|
+
timestamp: new Date().toISOString(),
|
|
99
|
+
},
|
|
100
|
+
applied: false,
|
|
101
|
+
phase: 'failed',
|
|
102
|
+
error: convergenceError,
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
73
106
|
sandbox.createWorktree()
|
|
74
107
|
|
|
75
108
|
const applied = sandbox.applyModification({
|
|
@@ -336,7 +336,7 @@ export async function selectTargetSkill(
|
|
|
336
336
|
return extractFilePath(response, skillFiles)
|
|
337
337
|
}
|
|
338
338
|
|
|
339
|
-
const PROSE_GENERATE_PROMPT_VERSION = '1.
|
|
339
|
+
const PROSE_GENERATE_PROMPT_VERSION = '1.1.0'
|
|
340
340
|
|
|
341
341
|
function buildGenerateProsePrompt(
|
|
342
342
|
signal: CrsiSignal,
|
|
@@ -357,7 +357,12 @@ function buildGenerateProsePrompt(
|
|
|
357
357
|
'当前内容:',
|
|
358
358
|
originalContent,
|
|
359
359
|
'',
|
|
360
|
-
'
|
|
360
|
+
'返回格式(严格遵守,两段):',
|
|
361
|
+
'第 1 行:一行 JSON,写下你对这次改动的**预期效果**与**风险**:',
|
|
362
|
+
'{"expectedDelta": <number 或 null>, "risk": "<字符串>"}',
|
|
363
|
+
'- expectedDelta 是预期该 skill 的任务表现提升**点数**(可正可负;无法预测写 null)。',
|
|
364
|
+
'- risk 是这次改动可能在哪方面变差(一句话)。',
|
|
365
|
+
'第 2 行起:改进后的完整 markdown(保持 YAML frontmatter 的 name/description 字段,正文针对失败信号做针对性改进)。不要用代码围栏包住。',
|
|
361
366
|
].join('\n')
|
|
362
367
|
}
|
|
363
368
|
|
|
@@ -366,16 +371,68 @@ function stripMarkdownFence(text: string): string {
|
|
|
366
371
|
return match ? match[1]! : text
|
|
367
372
|
}
|
|
368
373
|
|
|
374
|
+
/** prose 提议的解析产物:正文 + 可选的事前预登记(ε 与风险声明)。 */
|
|
375
|
+
export interface ProsePrediction {
|
|
376
|
+
body: string
|
|
377
|
+
/** ε:事前写下的预期提升点数。缺席 = 模型没预测(含显式写 null)。 */
|
|
378
|
+
expectedEffect?: number
|
|
379
|
+
/** R:风险声明。缺席 = 未声明。 */
|
|
380
|
+
risk?: string
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* 解析 prose 响应:可选的一行 JSON 前缀(ε)+ 正文。
|
|
385
|
+
*
|
|
386
|
+
* 顺序是**先归一化、后嗅探**(不可颠倒):stripMarkdownFence 的正则锚在串首
|
|
387
|
+
* (/^```(?:markdown|md)?\s*\n…\n```\s*$/)。若先剥「首行围栏」再嗅探,正文尾部的
|
|
388
|
+
* 那个 ``` 就再没有东西去剥它 ⇒ 孤立的尾部围栏会进入写盘路径。
|
|
389
|
+
*
|
|
390
|
+
* 认领标记是**含 `expectedDelta` 键**(盖住 number 与显式 null 两种写法);
|
|
391
|
+
* 其余任何情况都走兜底 —— 正文 = 归一化后的原文,一字不改。
|
|
392
|
+
*/
|
|
393
|
+
export function parseProsePrediction(raw: string): ProsePrediction {
|
|
394
|
+
const stripped = stripMarkdownFence(raw)
|
|
395
|
+
const lines = stripped.split('\n')
|
|
396
|
+
const firstIdx = lines.findIndex((l) => l.trim() !== '')
|
|
397
|
+
if (firstIdx === -1) return { body: stripped }
|
|
398
|
+
|
|
399
|
+
let parsed: unknown
|
|
400
|
+
try {
|
|
401
|
+
parsed = JSON.parse(lines[firstIdx]!.trim())
|
|
402
|
+
} catch {
|
|
403
|
+
return { body: stripped } // 首行不是 JSON → 兜底
|
|
404
|
+
}
|
|
405
|
+
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
|
|
406
|
+
return { body: stripped }
|
|
407
|
+
}
|
|
408
|
+
const rec = parsed as { expectedDelta?: unknown; risk?: unknown }
|
|
409
|
+
if (!('expectedDelta' in rec)) return { body: stripped } // 不带 ε 的 JSON 不吃
|
|
410
|
+
|
|
411
|
+
const expectedEffect = typeof rec.expectedDelta === 'number' ? rec.expectedDelta : undefined
|
|
412
|
+
const risk = typeof rec.risk === 'string' ? rec.risk : undefined
|
|
413
|
+
// 剥掉 JSON 行本身 + 紧随其后的空行
|
|
414
|
+
const body = lines
|
|
415
|
+
.slice(firstIdx + 1)
|
|
416
|
+
.join('\n')
|
|
417
|
+
.replace(/^[ \t]*\n/, '')
|
|
418
|
+
|
|
419
|
+
return {
|
|
420
|
+
body,
|
|
421
|
+
...(expectedEffect !== undefined ? { expectedEffect } : {}),
|
|
422
|
+
...(risk !== undefined ? { risk } : {}),
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
|
|
369
426
|
export async function generateProseContent(
|
|
370
427
|
signal: CrsiSignal,
|
|
371
428
|
llm: Llm,
|
|
372
429
|
filePath: string,
|
|
373
430
|
originalContent: string,
|
|
374
|
-
): Promise<
|
|
431
|
+
): Promise<ProsePrediction | null> {
|
|
375
432
|
const prompt = buildGenerateProsePrompt(signal, filePath, originalContent)
|
|
376
433
|
const response = await collectLlmText(llm, prompt)
|
|
377
434
|
if (!response) return null
|
|
378
|
-
return
|
|
435
|
+
return parseProsePrediction(response)
|
|
379
436
|
}
|
|
380
437
|
|
|
381
438
|
export interface ProseProposalResult {
|
|
@@ -383,6 +440,10 @@ export interface ProseProposalResult {
|
|
|
383
440
|
newContent: string
|
|
384
441
|
originalContent: string
|
|
385
442
|
description: string
|
|
443
|
+
/** ε:由模型在正文之前写下(见 parseProsePrediction)。 */
|
|
444
|
+
expectedEffect?: number
|
|
445
|
+
/** R:风险声明。 */
|
|
446
|
+
risk?: string
|
|
386
447
|
}
|
|
387
448
|
|
|
388
449
|
export async function produceProseProposal(
|
|
@@ -401,10 +462,17 @@ export async function produceProseProposal(
|
|
|
401
462
|
return null
|
|
402
463
|
}
|
|
403
464
|
|
|
404
|
-
const
|
|
405
|
-
if (!
|
|
465
|
+
const generated = await generateProseContent(signal, llm, filePath, originalContent)
|
|
466
|
+
if (!generated || !generated.body) return null
|
|
406
467
|
|
|
407
|
-
return {
|
|
468
|
+
return {
|
|
469
|
+
filePath,
|
|
470
|
+
newContent: generated.body,
|
|
471
|
+
originalContent,
|
|
472
|
+
description: signal.title,
|
|
473
|
+
...(generated.expectedEffect !== undefined ? { expectedEffect: generated.expectedEffect } : {}),
|
|
474
|
+
...(generated.risk !== undefined ? { risk: generated.risk } : {}),
|
|
475
|
+
}
|
|
408
476
|
}
|
|
409
477
|
|
|
410
478
|
const SKILL_DIRS: Array<[string, string]> = [
|
|
@@ -582,6 +650,7 @@ export async function produceCrossoverProposal(
|
|
|
582
650
|
newContent: string
|
|
583
651
|
originalContent: string
|
|
584
652
|
blastRadius: string[]
|
|
653
|
+
merge: boolean
|
|
585
654
|
} | null> {
|
|
586
655
|
const response = await collectLlmText(llm, buildCrossoverPrompt(currentLessons))
|
|
587
656
|
if (!response) return null
|
|
@@ -607,5 +676,6 @@ export async function produceCrossoverProposal(
|
|
|
607
676
|
newContent,
|
|
608
677
|
originalContent: currentLessons,
|
|
609
678
|
blastRadius: [LESSONS_FILE],
|
|
679
|
+
merge: true,
|
|
610
680
|
}
|
|
611
681
|
}
|
package/src/core/crsi-sandbox.ts
CHANGED
|
@@ -19,6 +19,7 @@ import { mkdirSync, rmSync, existsSync, writeFileSync, readFileSync, readdirSync
|
|
|
19
19
|
import { join, resolve, sep, posix } from 'node:path'
|
|
20
20
|
import { tmpdir, homedir } from 'node:os'
|
|
21
21
|
import { randomUUID } from 'node:crypto'
|
|
22
|
+
import { LESSONS_FILE, MANAGED_RULES_FILE } from './crsi-producer'
|
|
22
23
|
|
|
23
24
|
// ── Types ──
|
|
24
25
|
|
|
@@ -195,6 +196,70 @@ export function validateBlastRadius(proposal: {
|
|
|
195
196
|
return null
|
|
196
197
|
}
|
|
197
198
|
|
|
199
|
+
/**
|
|
200
|
+
* 脚手架三项计数(B_H 的度量)。按 filePath 分派语义单位:
|
|
201
|
+
* 教训段数 / 受管理规则条数 / 其余按 UTF-8 字节数。
|
|
202
|
+
*
|
|
203
|
+
* **为什么教训/规则文件不计字节**:合并会重写散文,字节数随措辞涨落。把字节计入,
|
|
204
|
+
* 会让「删二增一」因新写的合并段比原来两段更长而被误拦 —— 即闸会挡掉它本该允许的那件事。
|
|
205
|
+
* skill 文件没有可用的语义单位(它的「条数」就是文件本身),才退到字节数。
|
|
206
|
+
*
|
|
207
|
+
* `## ` 的口径与 `removeLessonSections` 逐字一致 —— 闸数的必须是 crossover 真正删得掉的那些
|
|
208
|
+
* 单位,否则两把尺子会各说各话。**刻意不声称与 `extractCrsiLessonSummaries` 一致**:后者用
|
|
209
|
+
* `/^##\s+(.+?)\s*$/`,还认 `##\t`,而 `startsWith('## ')` 不认(`'##\ta: 1'` 在此计 0、
|
|
210
|
+
* 在那里计 1)。闸依赖的是「删得掉」,故按前者对齐;这个差是选择,不是遗漏。
|
|
211
|
+
*
|
|
212
|
+
* 分派按**解析后**的路径(`resolve` 两侧同调,`cwd` 相消)—— 字面量比较时,`./` 前缀或
|
|
213
|
+
* 绝对形式的教训路径会静默落到**字节**分支,而那正是上面说绝不该用在教训文件上的那把尺子。
|
|
214
|
+
* 兄弟守卫 `isProtectedPath` 本身不做规范化(纯前缀比较);是调用点 `CrsiSandbox.applyModification`
|
|
215
|
+
* 先 `posix.normalize` 再调它(`proposal-guard.ts` 那条调用点未规范化)。
|
|
216
|
+
*/
|
|
217
|
+
export function measureScaffold(
|
|
218
|
+
filePath: string,
|
|
219
|
+
content: string,
|
|
220
|
+
): { lessons: number; rules: number; bytes: number } {
|
|
221
|
+
if (resolve(filePath) === resolve(LESSONS_FILE)) {
|
|
222
|
+
const lessons = content.split('\n').filter((l) => l.startsWith('## ')).length
|
|
223
|
+
return { lessons, rules: 0, bytes: 0 }
|
|
224
|
+
}
|
|
225
|
+
if (resolve(filePath) === resolve(MANAGED_RULES_FILE)) {
|
|
226
|
+
const rules = (content.match(/id: '/g) ?? []).length
|
|
227
|
+
return { lessons: 0, rules, bytes: 0 }
|
|
228
|
+
}
|
|
229
|
+
return { lessons: 0, rules: 0, bytes: Buffer.byteLength(content, 'utf-8') }
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* 合并型提案的收敛闸(B_H)。**零写死常数** —— 它不设上界,只要求「合并这件事本身别把
|
|
234
|
+
* 脚手架抬高」。RSIH 的 ‖H‖≤B_H 需要一个数,是因为它必须允许学到上界、到顶再强制合并;
|
|
235
|
+
* 本仓已有 dedup 那一半(教训按 `## category: title` 幂等、规则按 id 幂等),
|
|
236
|
+
* 缺的只是另一半,而那一半不需要数:**闸只在「试图整合」那一刻开火**。
|
|
237
|
+
*
|
|
238
|
+
* 返回拒绝理由,合法时返回 null(同 validateBlastRadius 的签名)。
|
|
239
|
+
*/
|
|
240
|
+
export function validateMergeConvergence(proposal: {
|
|
241
|
+
filePath?: string
|
|
242
|
+
originalContent?: string
|
|
243
|
+
newContent?: string
|
|
244
|
+
merge?: boolean
|
|
245
|
+
}): string | null {
|
|
246
|
+
if (proposal.merge !== true) return null
|
|
247
|
+
// 无基线不是有增长:手工路径在文件不存在时正是这个形态(commands.ts 的宽松模式)。
|
|
248
|
+
if (!proposal.originalContent || !proposal.newContent) return null
|
|
249
|
+
|
|
250
|
+
const filePath = proposal.filePath ?? ''
|
|
251
|
+
const before = measureScaffold(filePath, proposal.originalContent)
|
|
252
|
+
const after = measureScaffold(filePath, proposal.newContent)
|
|
253
|
+
|
|
254
|
+
const rose: string[] = []
|
|
255
|
+
if (after.lessons > before.lessons) rose.push(`教训段数 ${before.lessons} → ${after.lessons}`)
|
|
256
|
+
if (after.rules > before.rules) rose.push(`规则条数 ${before.rules} → ${after.rules}`)
|
|
257
|
+
if (after.bytes > before.bytes) rose.push(`字节数 ${before.bytes} → ${after.bytes}`)
|
|
258
|
+
if (rose.length === 0) return null
|
|
259
|
+
|
|
260
|
+
return `合并型提案必须收敛,但脚手架增长了:${rose.join(';')}。`
|
|
261
|
+
}
|
|
262
|
+
|
|
198
263
|
// ── Sandbox ──
|
|
199
264
|
|
|
200
265
|
export class CrsiSandbox {
|
package/src/core/eval-harness.ts
CHANGED
|
@@ -21,14 +21,21 @@ import { PreFlightChecker } from './preflight-checker'
|
|
|
21
21
|
import { createDefaultPostFlightChecker } from './post-flight-checker'
|
|
22
22
|
import { WorkingMemory } from './working-memory'
|
|
23
23
|
import { RedTeam } from './red-team'
|
|
24
|
-
import {
|
|
24
|
+
import {
|
|
25
|
+
isProtectedPath,
|
|
26
|
+
validateBlastRadius,
|
|
27
|
+
validateMergeConvergence,
|
|
28
|
+
PROTECTED_CRITICAL_FILES,
|
|
29
|
+
} from './crsi-sandbox'
|
|
25
30
|
import {
|
|
26
31
|
produceRuleProposal,
|
|
27
32
|
MANAGED_RULES_FILE,
|
|
33
|
+
LESSONS_FILE,
|
|
28
34
|
buildLessonContent,
|
|
29
35
|
renderManagedRuleSource,
|
|
30
36
|
} from './crsi-producer'
|
|
31
37
|
import type { CrsiSignal } from './crsi-producer'
|
|
38
|
+
import { predictionHit } from './improvement-track'
|
|
32
39
|
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
33
40
|
|
|
34
41
|
// ── Types ──
|
|
@@ -43,6 +50,12 @@ export interface EvalResult {
|
|
|
43
50
|
detail?: string
|
|
44
51
|
/** 契约角色。缺省 neutral。 */
|
|
45
52
|
role?: ContractRole
|
|
53
|
+
/**
|
|
54
|
+
* anchor 契约的**定义处真源**。`role` 由此派生(见 runEval 末尾的回填)。
|
|
55
|
+
* `ANCHOR_CONTRACT_IDS` 退为**独立声明**,两者由 test/integrity/anchor-contract-wiring
|
|
56
|
+
* 守卫两向比对 —— 两处独立陈述同一件事,它们才可能不一致,守卫才有内容。
|
|
57
|
+
*/
|
|
58
|
+
anchor?: true
|
|
46
59
|
}
|
|
47
60
|
|
|
48
61
|
export interface EvalReport {
|
|
@@ -70,6 +83,8 @@ export const ANCHOR_CONTRACT_IDS: ReadonlySet<string> = new Set([
|
|
|
70
83
|
'red-team-zero-gaps',
|
|
71
84
|
'producer-rule-shape',
|
|
72
85
|
'producer-rule-idempotent',
|
|
86
|
+
'prediction-hit-truth-table',
|
|
87
|
+
'merge-convergence-gate',
|
|
73
88
|
'self-report-diagnostic',
|
|
74
89
|
])
|
|
75
90
|
|
|
@@ -147,6 +162,7 @@ export function runEval(): EvalReport {
|
|
|
147
162
|
id: 'rule-timeout',
|
|
148
163
|
description: '内置 timeout 规则命中低超时的 npm install',
|
|
149
164
|
passed: timeout.modified.timeout === 300000,
|
|
165
|
+
anchor: true,
|
|
150
166
|
})
|
|
151
167
|
|
|
152
168
|
const gitForce = ruleEngine.intercept('Bash', {
|
|
@@ -157,6 +173,7 @@ export function runEval(): EvalReport {
|
|
|
157
173
|
id: 'rule-git-force',
|
|
158
174
|
description: 'git --force 触发告警',
|
|
159
175
|
passed: gitForce.warnings.length > 0,
|
|
176
|
+
anchor: true,
|
|
160
177
|
})
|
|
161
178
|
|
|
162
179
|
const disabledRule: import('./rule-engine').ToolRule = {
|
|
@@ -174,6 +191,7 @@ export function runEval(): EvalReport {
|
|
|
174
191
|
id: 'rule-disabled-skip',
|
|
175
192
|
description: '禁用规则被跳过',
|
|
176
193
|
passed: disabled.warnings.length === 0,
|
|
194
|
+
anchor: true,
|
|
177
195
|
})
|
|
178
196
|
|
|
179
197
|
// ── 宪法(ground truth:8 原则 + facet 映射 + 愿力序言) ──
|
|
@@ -182,6 +200,7 @@ export function runEval(): EvalReport {
|
|
|
182
200
|
id: 'constitution-8-principles',
|
|
183
201
|
description: '宪法含 8 条原则',
|
|
184
202
|
passed: principles.length === 8,
|
|
203
|
+
anchor: true,
|
|
185
204
|
})
|
|
186
205
|
|
|
187
206
|
const prajna = principles.filter((p) => p.facet === 'prajna').length
|
|
@@ -191,12 +210,14 @@ export function runEval(): EvalReport {
|
|
|
191
210
|
id: 'constitution-facets',
|
|
192
211
|
description: 'facet 映射 智3 / 金刚5 / 悲0',
|
|
193
212
|
passed: prajna === 3 && vajra === 5 && karuna === 0,
|
|
213
|
+
anchor: true,
|
|
194
214
|
})
|
|
195
215
|
|
|
196
216
|
results.push({
|
|
197
217
|
id: 'constitution-preamble',
|
|
198
218
|
description: '愿力序言已注入',
|
|
199
219
|
passed: !!DEFAULT_CONSTITUTION.preamble && DEFAULT_CONSTITUTION.preamble.includes('悲'),
|
|
220
|
+
anchor: true,
|
|
200
221
|
})
|
|
201
222
|
|
|
202
223
|
// ── 沙箱只读边界(ground truth:受保护路径被拒) ──
|
|
@@ -206,7 +227,12 @@ export function runEval(): EvalReport {
|
|
|
206
227
|
['sandbox-protected-machinery', 'apps/cli/src/core/crsi-sandbox.ts'],
|
|
207
228
|
]
|
|
208
229
|
for (const [id, path] of protectedChecks) {
|
|
209
|
-
results.push({
|
|
230
|
+
results.push({
|
|
231
|
+
id,
|
|
232
|
+
description: `受保护路径被拒: ${path}`,
|
|
233
|
+
passed: isProtectedPath(path),
|
|
234
|
+
anchor: true,
|
|
235
|
+
})
|
|
210
236
|
}
|
|
211
237
|
|
|
212
238
|
// ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
|
|
@@ -216,6 +242,7 @@ export function runEval(): EvalReport {
|
|
|
216
242
|
description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
|
|
217
243
|
passed: unprotected.length === 0,
|
|
218
244
|
...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
|
|
245
|
+
anchor: true,
|
|
219
246
|
})
|
|
220
247
|
|
|
221
248
|
// ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
|
|
@@ -226,6 +253,7 @@ export function runEval(): EvalReport {
|
|
|
226
253
|
validateBlastRadius({ blastRadius: undefined }) !== null &&
|
|
227
254
|
validateBlastRadius({ blastRadius: [] }) !== null &&
|
|
228
255
|
validateBlastRadius({ blastRadius: ['apps/cli/src/foo.ts'] }) === null,
|
|
256
|
+
anchor: true,
|
|
229
257
|
})
|
|
230
258
|
|
|
231
259
|
// ── 安全(ground truth:16 攻击零漏过) ──
|
|
@@ -235,6 +263,7 @@ export function runEval(): EvalReport {
|
|
|
235
263
|
description: '16 个对抗场景零漏过',
|
|
236
264
|
passed: redTeam.passedThrough === 0,
|
|
237
265
|
detail: `score=${redTeam.score}, passedThrough=${redTeam.passedThrough}, falsePositives=${redTeam.falsePositives}`,
|
|
266
|
+
anchor: true,
|
|
238
267
|
})
|
|
239
268
|
|
|
240
269
|
// ── producer 行为(ground truth:固化规则产出正确 shape + 幂等) ──
|
|
@@ -255,6 +284,7 @@ export function runEval(): EvalReport {
|
|
|
255
284
|
ruleProposal.newContent.includes("source: 'managed'") &&
|
|
256
285
|
ruleProposal.newContent.includes('timeout: 300000') &&
|
|
257
286
|
ruleProposal.newContent.includes('enabled: true'),
|
|
287
|
+
anchor: true,
|
|
258
288
|
})
|
|
259
289
|
|
|
260
290
|
results.push({
|
|
@@ -262,6 +292,7 @@ export function runEval(): EvalReport {
|
|
|
262
292
|
description: '同名规则重复产出被拒(幂等)',
|
|
263
293
|
passed:
|
|
264
294
|
ruleProposal !== null && produceRuleProposal(frozenSignal, ruleProposal.newContent) === null,
|
|
295
|
+
anchor: true,
|
|
265
296
|
})
|
|
266
297
|
|
|
267
298
|
// ── 组件归因(ground truth:缺省 experiential、显式组件透传、非 experiential 不进 managed-rule) ──
|
|
@@ -334,6 +365,46 @@ export function runEval(): EvalReport {
|
|
|
334
365
|
results.push({ ...judgeBehaviorTask(task, ruleEngine), role: 'target' })
|
|
335
366
|
}
|
|
336
367
|
|
|
368
|
+
// ── ε 预测命中真值表(ground truth:命中判据不叠加统计阈值) ──
|
|
369
|
+
// `(20, 20)` 那条**承重**:判据是 `deltaMean >= predicted` 而 `>=` 与 `>` 只在
|
|
370
|
+
// `predicted === deltaMean` 处分歧 ⇒ 少了它,「把 >= 翻成 >」在契约上不可观测。
|
|
371
|
+
results.push({
|
|
372
|
+
id: 'prediction-hit-truth-table',
|
|
373
|
+
description: 'predictionHit 真值表(返回值):未达不算、达到或恰好相等算命中、缺席恒 false',
|
|
374
|
+
passed:
|
|
375
|
+
predictionHit(50, 20) === false &&
|
|
376
|
+
predictionHit(10, 20) === true &&
|
|
377
|
+
predictionHit(20, 20) === true &&
|
|
378
|
+
predictionHit(undefined, 20) === false,
|
|
379
|
+
anchor: true,
|
|
380
|
+
})
|
|
381
|
+
|
|
382
|
+
// ── B_H 合并型收敛闸(ground truth:净增被拒、删二增一通过、非合并型不受此闸) ──
|
|
383
|
+
results.push({
|
|
384
|
+
id: 'merge-convergence-gate',
|
|
385
|
+
description: '合并型净增被拒、删二增一通过、merge=false 净增通过',
|
|
386
|
+
passed:
|
|
387
|
+
validateMergeConvergence({
|
|
388
|
+
filePath: LESSONS_FILE,
|
|
389
|
+
originalContent: '## a: 1\n\n## b: 2\n',
|
|
390
|
+
newContent: '## a: 1\n\n## b: 2\n\n## c: 3\n',
|
|
391
|
+
merge: true,
|
|
392
|
+
}) !== null &&
|
|
393
|
+
validateMergeConvergence({
|
|
394
|
+
filePath: LESSONS_FILE,
|
|
395
|
+
originalContent: '## a: 1\n\n## b: 2\n',
|
|
396
|
+
newContent: '## ab: merged\n',
|
|
397
|
+
merge: true,
|
|
398
|
+
}) === null &&
|
|
399
|
+
validateMergeConvergence({
|
|
400
|
+
filePath: LESSONS_FILE,
|
|
401
|
+
originalContent: '## a: 1\n',
|
|
402
|
+
newContent: '## a: 1\n\n## b: 2\n',
|
|
403
|
+
merge: false,
|
|
404
|
+
}) === null,
|
|
405
|
+
anchor: true,
|
|
406
|
+
})
|
|
407
|
+
|
|
337
408
|
// ── 自报分数只作诊断:评分路径无 LLM,分数来自 ground-truth 契约而非模型自报 ──
|
|
338
409
|
// anchor 锁死「评分组件不暴露 LLM 的 chat 能力」。4 个组件(ruleEngine/constitution/
|
|
339
410
|
// errorDB/preflight)都是确定性组件(runEval 同步评分)。若未来有人把 LLM 注入评分
|
|
@@ -347,11 +418,13 @@ export function runEval(): EvalReport {
|
|
|
347
418
|
id: 'self-report-diagnostic',
|
|
348
419
|
description: '评分无 LLM:机制哨兵组件不暴露 chat 能力(分数只来自 ground-truth,非模型自报)',
|
|
349
420
|
passed: !llmInjected,
|
|
421
|
+
anchor: true,
|
|
350
422
|
})
|
|
351
423
|
|
|
352
|
-
// 角色标注:anchor
|
|
424
|
+
// 角色标注:anchor 由契约**定义处内联的标记**派生(真源),
|
|
425
|
+
// ANCHOR_CONTRACT_IDS 退为独立声明 —— 两者由 anchor-contract-wiring 守卫两向比对。
|
|
353
426
|
for (const r of results) {
|
|
354
|
-
if (
|
|
427
|
+
if (r.anchor) r.role = 'anchor'
|
|
355
428
|
}
|
|
356
429
|
|
|
357
430
|
// anchor 自检(ground truth:所有 anchor 契约必须全绿,否则门拒)。
|
|
@@ -22,6 +22,10 @@ export interface ImprovementReport {
|
|
|
22
22
|
noise: number
|
|
23
23
|
minEffect: number
|
|
24
24
|
verdict: ImprovementVerdict
|
|
25
|
+
/** ε:提交者事前写下的预期提升点数。缺席 = 该记录没有预登记。 */
|
|
26
|
+
predictedDelta?: number
|
|
27
|
+
/** 预测是否命中。与 predictedDelta 同时出现、同时缺席(JSON 序列化会丢掉 undefined 键)。 */
|
|
28
|
+
predictionHit?: boolean
|
|
25
29
|
}
|
|
26
30
|
|
|
27
31
|
export interface ImprovementRecord extends ImprovementReport {
|
|
@@ -55,6 +59,7 @@ function stdDev(xs: number[]): number {
|
|
|
55
59
|
export function buildImprovementReport(
|
|
56
60
|
sample: SkillDeltaSample,
|
|
57
61
|
changeSet: string[],
|
|
62
|
+
predicted?: number,
|
|
58
63
|
): ImprovementReport {
|
|
59
64
|
const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
|
|
60
65
|
const noise = stdDev(sample.baselineScores)
|
|
@@ -70,6 +75,14 @@ export function buildImprovementReport(
|
|
|
70
75
|
noise,
|
|
71
76
|
minEffect,
|
|
72
77
|
verdict,
|
|
78
|
+
// 两个字段同生同灭:缺席预测必须**键不存在**,而不是 predictionHit: false。
|
|
79
|
+
// 理由**不是**「predictionHit: false 会被算进分母」—— 分母只认 `predictedDelta !== undefined`
|
|
80
|
+
//(只写 `predictionHit: false` 而不写 `predictedDelta` 会被整条忽略)。真正的风险是**反过来的半条**:
|
|
81
|
+
// 有 `predictedDelta` 而无 `predictionHit` ⇒ 该条进了分母,却永远不可能被算成命中
|
|
82
|
+
//(分子只认 `predictionHit === true`),等于一条静默的「未命中」。
|
|
83
|
+
...(predicted !== undefined
|
|
84
|
+
? { predictedDelta: predicted, predictionHit: predictionHit(predicted, deltaMean) }
|
|
85
|
+
: {}),
|
|
73
86
|
}
|
|
74
87
|
}
|
|
75
88
|
|
|
@@ -107,6 +120,34 @@ export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
|
|
|
107
120
|
return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
|
|
108
121
|
}
|
|
109
122
|
|
|
123
|
+
/**
|
|
124
|
+
* 预测命中:事前写下的点数被实际达到。缺席预测(undefined)不计入。
|
|
125
|
+
*
|
|
126
|
+
* 刻意**不叠加 `minEffect`** —— ε 是提交者自己写下的数,判据就是「达到没达到」;
|
|
127
|
+
* 再套一层统计阈值会让两个数打架,且使「命中」不可复算。
|
|
128
|
+
*/
|
|
129
|
+
export function predictionHit(predicted: number | undefined, deltaMean: number): boolean {
|
|
130
|
+
return predicted !== undefined && deltaMean >= predicted
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* ε 命中率。分母 = **有预测的记录数**(判据取 `predictedDelta !== undefined`),
|
|
135
|
+
* 复用既有 wilsonInterval。无预测的记录既不入分子也不入分母。
|
|
136
|
+
*/
|
|
137
|
+
export function predictionHitRate(records: ImprovementRecord[]): {
|
|
138
|
+
total: number
|
|
139
|
+
hits: number
|
|
140
|
+
rate: number
|
|
141
|
+
lo: number
|
|
142
|
+
hi: number
|
|
143
|
+
} {
|
|
144
|
+
const judged = records.filter((r) => r.predictedDelta !== undefined)
|
|
145
|
+
const total = judged.length
|
|
146
|
+
const hits = judged.filter((r) => r.predictionHit === true).length
|
|
147
|
+
const { lo, hi } = wilsonInterval(hits, total)
|
|
148
|
+
return { total, hits, rate: total === 0 ? 0 : hits / total, lo, hi }
|
|
149
|
+
}
|
|
150
|
+
|
|
110
151
|
// ── 台账 ──
|
|
111
152
|
|
|
112
153
|
export function improvementPath(): string {
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
export const PACKAGE_NAME = '@miphamai/cli' as const
|
|
10
10
|
|
|
11
11
|
/** 当前发布版本 */
|
|
12
|
-
export const PACKAGE_VERSION = '0.
|
|
12
|
+
export const PACKAGE_VERSION = '0.82.0' as const
|
|
13
13
|
|
|
14
14
|
/** npm install 全局安装命令 */
|
|
15
15
|
export const NPM_INSTALL_COMMAND = `npm install -g ${PACKAGE_NAME}` as const
|
package/src/ui/commands.ts
CHANGED
|
@@ -50,6 +50,7 @@ import {
|
|
|
50
50
|
appendImprovement,
|
|
51
51
|
readImprovements,
|
|
52
52
|
improvementRate,
|
|
53
|
+
predictionHitRate,
|
|
53
54
|
setPendingVerdict,
|
|
54
55
|
getPendingVerdict,
|
|
55
56
|
shouldBlockApproval,
|
|
@@ -804,6 +805,25 @@ const crsiStatsCmd: CommandHandler = async (ctx) => {
|
|
|
804
805
|
}
|
|
805
806
|
}
|
|
806
807
|
|
|
808
|
+
// ── ε 预测命中(prose 路径) ──
|
|
809
|
+
// 作废条款:样本不足就不下结论;台账攒够 20 条而判定样本仍 < 5 ⇒ 明写机制失效。
|
|
810
|
+
// 只打印、不写台账:本命令是只读的,而该结论每次都能从同一份 improvements.jsonl 重算出来。
|
|
811
|
+
const records = readImprovements()
|
|
812
|
+
const pred = predictionHitRate(records)
|
|
813
|
+
lines.push('')
|
|
814
|
+
lines.push('### ε 预测命中(prose 路径)')
|
|
815
|
+
if (pred.total < 5) {
|
|
816
|
+
lines.push(`样本不足(判定记录 ${pred.total} 条,需 ≥ 5)—— 不下结论。`)
|
|
817
|
+
if (records.length >= 20) {
|
|
818
|
+
lines.push('⚠️ ε 机制失效:prose 路径使用率过低(记录总数已达 20 而判定样本仍 < 5)。')
|
|
819
|
+
}
|
|
820
|
+
} else {
|
|
821
|
+
lines.push(
|
|
822
|
+
`命中率: ${pred.hits}/${pred.total} (${(pred.rate * 100).toFixed(0)}%, ` +
|
|
823
|
+
`Wilson 95% [${(pred.lo * 100).toFixed(0)}%, ${(pred.hi * 100).toFixed(0)}%])`,
|
|
824
|
+
)
|
|
825
|
+
}
|
|
826
|
+
|
|
807
827
|
return { content: lines.join('\n') }
|
|
808
828
|
}
|
|
809
829
|
|
|
@@ -970,11 +990,36 @@ const crsiProposeCmd: CommandHandler = async (ctx, args) => {
|
|
|
970
990
|
return { content: `❌ 生成失败(phase: ${result.phase})。\n${result.error ?? ''}` }
|
|
971
991
|
}
|
|
972
992
|
|
|
993
|
+
// ε 预登记落地:prose 路径此前**不测量**(measureSkillDeltaRepeated 全仓库只有手工路径一个
|
|
994
|
+
// 调用点)⇒ ε 曾在 A 流程登记、判定侧在 B 流程,两端永不相遇。这里补上测量。
|
|
995
|
+
// 成本:每次提案多 6 次 LLM 调用(同手工路径 :867 的注释)。
|
|
996
|
+
let predictionLine = ''
|
|
997
|
+
try {
|
|
998
|
+
const sample = await measureSkillDeltaRepeated(llm, {
|
|
999
|
+
filePath: proposal.filePath,
|
|
1000
|
+
originalContent: proposal.originalContent,
|
|
1001
|
+
newContent: proposal.newContent,
|
|
1002
|
+
})
|
|
1003
|
+
if (sample) {
|
|
1004
|
+
const report = buildImprovementReport(sample, [proposal.filePath], proposal.expectedEffect)
|
|
1005
|
+
setPendingVerdict(report.verdict)
|
|
1006
|
+
appendImprovement({ ...report, id: randomUUID(), timestamp: new Date().toISOString() })
|
|
1007
|
+
if (report.predictionHit !== undefined) {
|
|
1008
|
+
predictionLine =
|
|
1009
|
+
`\n🎯 ε 预测命中: ${report.predictionHit ? '命中 ✅' : '未命中 ⚠️'}` +
|
|
1010
|
+
`(预测 ${report.predictedDelta},实际 delta ${report.deltaMean.toFixed(1)})`
|
|
1011
|
+
}
|
|
1012
|
+
}
|
|
1013
|
+
} catch {
|
|
1014
|
+
// 测量失败(LLM 不可用等)不阻断提案流程 —— 与手工路径 :889 的处置一致。
|
|
1015
|
+
}
|
|
1016
|
+
|
|
973
1017
|
appendProseProposal({ id, filePath: proposal.filePath, timestamp: new Date().toISOString() })
|
|
974
1018
|
|
|
975
1019
|
return {
|
|
976
1020
|
content:
|
|
977
1021
|
`✅ 已生成散文提议并跑过测试。审阅 diff:\n\n${result.diff}\n\n` +
|
|
1022
|
+
predictionLine +
|
|
978
1023
|
'/crsi modify --approve 合并 | /crsi modify --reject 丢弃',
|
|
979
1024
|
}
|
|
980
1025
|
}
|