@miphamai/cli 0.63.0 → 0.65.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/standard/safe-coding.SKILL.md +9 -0
- package/src/config/defaults.ts +4 -0
- package/src/core/autocomplete.ts +64 -0
- package/src/core/crsi-modify.ts +13 -11
- package/src/core/crsi-producer.ts +128 -2
- package/src/core/crsi-sandbox.ts +60 -22
- package/src/core/eval-harness.ts +23 -8
- package/src/core/improvement-track.ts +144 -0
- package/src/core/instructions.ts +12 -0
- package/src/core/reward-fn.ts +44 -0
- package/src/core/task-performance-tasks.json +42 -0
- package/src/core/task-performance.ts +244 -0
- package/src/i18n-core/locales/en-US.json +3 -1
- package/src/i18n-core/locales/zh-CN.json +3 -1
- package/src/shared/package-info.ts +1 -1
- package/src/shared/types.ts +7 -0
- package/src/shared/update.ts +41 -0
- package/src/skills/bundled-skills.ts +1 -0
- package/src/skills/loader.ts +1 -1
- package/src/ui/agent-footer.tsx +5 -1
- package/src/ui/app.tsx +71 -3
- package/src/ui/commands.ts +162 -8
- package/src/ui/input.tsx +71 -0
package/package.json
CHANGED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: safe-coding
|
|
3
|
+
description: Safe coding rules for code generation — validate external/user input before use and throw RangeError on invalid input
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Safe Coding
|
|
8
|
+
|
|
9
|
+
处理外部/用户输入前必须校验:`null`、`undefined`、空字符串、格式非法时,抛出 `RangeError`,消息为 `'invalid input'`。
|
package/src/config/defaults.ts
CHANGED
|
@@ -16,6 +16,10 @@ export const DEFAULT_CONFIG: MiphamConfig = {
|
|
|
16
16
|
showThinking: 'off',
|
|
17
17
|
showSchedulingNotices: false,
|
|
18
18
|
showCommandPicker: false,
|
|
19
|
+
autocomplete: {
|
|
20
|
+
enabled: true,
|
|
21
|
+
debounceMs: 400,
|
|
22
|
+
},
|
|
19
23
|
// Org 级权限限制(可选):forbiddenModes 禁指定模式 / maxAllowedMode 封顶层级;
|
|
20
24
|
// 请求被禁模式时 fail-closed 降级(如 forbiddenModes:['bypassPermissions'])。
|
|
21
25
|
// permissionRestrictions: { forbiddenModes: ['bypassPermissions'] },
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { ChatRequest } from '../providers/registry'
|
|
2
|
+
import type { Llm } from '../providers/llm'
|
|
3
|
+
|
|
4
|
+
export const AUTOCOMPLETE_SYSTEM_PROMPT =
|
|
5
|
+
'你是续写助手。只续写用户正在输入的这条消息,只返回续写部分(不要重复已输入的文字、不要解释、不要换行)。'
|
|
6
|
+
|
|
7
|
+
/** 带上最近几条对话(含待续写输入),供续写贴合上下文。 */
|
|
8
|
+
export const AUTOCOMPLETE_MAX_CONTEXT = 6
|
|
9
|
+
|
|
10
|
+
export interface RecentMessage {
|
|
11
|
+
role: 'user' | 'assistant'
|
|
12
|
+
content: string
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/** 拼续写请求:systemPrompt + 最近 N 条 + 当前输入作为待续写消息。 */
|
|
16
|
+
export function buildAutocompleteRequest(recent: RecentMessage[], input: string): ChatRequest {
|
|
17
|
+
return {
|
|
18
|
+
model: '', // falsy → registry 回退 active model
|
|
19
|
+
messages: [...recent.slice(-AUTOCOMPLETE_MAX_CONTEXT), { role: 'user', content: input }],
|
|
20
|
+
systemPrompt: AUTOCOMPLETE_SYSTEM_PROMPT,
|
|
21
|
+
temperature: 0,
|
|
22
|
+
maxTokens: 64,
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** 剥掉 LLM 可能重复的 input 前缀,返回纯续写 suffix;空/无效 → null。 */
|
|
27
|
+
export function extractCompletion(response: string, input: string): string | null {
|
|
28
|
+
let completion = response.trim()
|
|
29
|
+
if (!completion) return null
|
|
30
|
+
const normInput = input.trim()
|
|
31
|
+
if (normInput && completion.startsWith(normInput)) {
|
|
32
|
+
completion = completion.slice(normInput.length).trimStart()
|
|
33
|
+
}
|
|
34
|
+
return completion || null
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 触发 guard:非空、非 `/`·`@` 开头、非 loading、无活跃 picker。 */
|
|
38
|
+
export function shouldAutocomplete(
|
|
39
|
+
value: string,
|
|
40
|
+
isLoading: boolean,
|
|
41
|
+
pickerActive: boolean,
|
|
42
|
+
): boolean {
|
|
43
|
+
if (!value.trim()) return false
|
|
44
|
+
if (value.startsWith('/') || value.startsWith('@')) return false
|
|
45
|
+
if (isLoading) return false
|
|
46
|
+
if (pickerActive) return false
|
|
47
|
+
return true
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** 异步取建议:llm.chat → 竞态检查(isStale)→ 剥前缀。stale / 空 → null。 */
|
|
51
|
+
export async function requestSuggestion(
|
|
52
|
+
llm: Llm,
|
|
53
|
+
recent: RecentMessage[],
|
|
54
|
+
input: string,
|
|
55
|
+
isStale: () => boolean,
|
|
56
|
+
): Promise<string | null> {
|
|
57
|
+
const req = buildAutocompleteRequest(recent, input)
|
|
58
|
+
let text = ''
|
|
59
|
+
for await (const chunk of llm.chat(req)) {
|
|
60
|
+
if (chunk.type === 'text' && chunk.content) text += chunk.content
|
|
61
|
+
}
|
|
62
|
+
if (isStale()) return null
|
|
63
|
+
return extractCompletion(text, input)
|
|
64
|
+
}
|
package/src/core/crsi-modify.ts
CHANGED
|
@@ -15,7 +15,8 @@
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
16
|
import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
|
|
17
17
|
import type { CrsiModificationResult } from './crsi-sandbox'
|
|
18
|
-
import {
|
|
18
|
+
import { appendEvalScore, getLastEvalScore } from './eval-harness'
|
|
19
|
+
import { mechanismSentinel, type RewardFn } from './reward-fn'
|
|
19
20
|
|
|
20
21
|
export interface CrsiProposal {
|
|
21
22
|
/** 人类可读的改动说明 */
|
|
@@ -45,10 +46,11 @@ let pendingSandbox: CrsiSandbox | null = null
|
|
|
45
46
|
* 编排完整 5 阶段:createWorktree → applyModification → runTests →(失败自动
|
|
46
47
|
* rollback)→ getDiff。测试通过后暂存为 pending,返回 diff 供人类审阅。
|
|
47
48
|
*/
|
|
48
|
-
export function runCrsiModification(
|
|
49
|
+
export async function runCrsiModification(
|
|
49
50
|
proposal: CrsiProposal,
|
|
50
51
|
sandbox: CrsiSandbox = new CrsiSandbox(),
|
|
51
|
-
|
|
52
|
+
opts?: { rewardFn?: RewardFn },
|
|
53
|
+
): Promise<CrsiModificationResult> {
|
|
52
54
|
// 完整覆盖闸:自修改必须声明 blastRadius,否则 fail-closed(在 worktree 之前,零副作用)。
|
|
53
55
|
const blastRadiusError = validateBlastRadius(proposal)
|
|
54
56
|
if (blastRadiusError) {
|
|
@@ -96,18 +98,18 @@ export function runCrsiModification(
|
|
|
96
98
|
return applied
|
|
97
99
|
}
|
|
98
100
|
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
|
|
102
|
-
const
|
|
103
|
-
const last = getLastEvalScore()
|
|
104
|
-
if (last !== null &&
|
|
101
|
+
// Reward gate:奖励分数不得低于上次记录(防跨合并退化)。
|
|
102
|
+
// 默认机制哨兵;可插拔——调用方传 opts.rewardFn 换用其他奖励源(如任务表现)。
|
|
103
|
+
const rewardFn = opts?.rewardFn ?? mechanismSentinel()
|
|
104
|
+
const report = await rewardFn.evaluate()
|
|
105
|
+
const last = getLastEvalScore(rewardFn.name)
|
|
106
|
+
if (last !== null && report.score < last) {
|
|
105
107
|
sandbox.rollback()
|
|
106
108
|
applied.phase = 'failed'
|
|
107
|
-
applied.error = `
|
|
109
|
+
applied.error = `Reward regression (${rewardFn.name}): score ${report.score} < last ${last}`
|
|
108
110
|
return applied
|
|
109
111
|
}
|
|
110
|
-
appendEvalScore(
|
|
112
|
+
appendEvalScore(rewardFn.name, report)
|
|
111
113
|
|
|
112
114
|
applied.phase = 'passed'
|
|
113
115
|
applied.diff = sandbox.getDiff()
|
|
@@ -67,14 +67,18 @@ export function selectCrsiSignal(
|
|
|
67
67
|
}
|
|
68
68
|
|
|
69
69
|
/** 模板化地把信号渲染成一段教训 markdown(不动 LLM)。 */
|
|
70
|
-
export function buildLessonContent(
|
|
70
|
+
export function buildLessonContent(
|
|
71
|
+
signal: CrsiSignal,
|
|
72
|
+
timestamp: string,
|
|
73
|
+
source = 'CRSI producer (autoApplicable)',
|
|
74
|
+
): string {
|
|
71
75
|
const lines: string[] = [
|
|
72
76
|
`## ${signal.category}: ${signal.title}`,
|
|
73
77
|
'',
|
|
74
78
|
`- 建议: ${signal.suggestion}`,
|
|
75
79
|
]
|
|
76
80
|
if (signal.severity) lines.push(`- 严重度: ${signal.severity}`)
|
|
77
|
-
lines.push(`- 生成时间: ${timestamp}`,
|
|
81
|
+
lines.push(`- 生成时间: ${timestamp}`, `- 来源: ${source}`, '', '### 证据')
|
|
78
82
|
for (const e of signal.evidence) lines.push(`- ${e}`)
|
|
79
83
|
lines.push('')
|
|
80
84
|
return lines.join('\n')
|
|
@@ -455,3 +459,125 @@ export function clearProseProposals(): number {
|
|
|
455
459
|
return 0
|
|
456
460
|
}
|
|
457
461
|
}
|
|
462
|
+
|
|
463
|
+
// ── Producer Crossover(第 4 原子算子):合并两条重叠教训 ──
|
|
464
|
+
// LLM 只生成(选对 + 合并版),判定全走确定性 guard + 沙箱 gate。A1 不破:无 LLM 自评。
|
|
465
|
+
|
|
466
|
+
const CROSSOVER_PROMPT_VERSION = '1.0.0'
|
|
467
|
+
|
|
468
|
+
function buildCrossoverPrompt(currentLessons: string): string {
|
|
469
|
+
return [
|
|
470
|
+
`你是 CRSI producer(producer-crossover v${CROSSOVER_PROMPT_VERSION})。给定当前教训文件,找出两条主题重叠、可合并的教训,生成一条综合教训。`,
|
|
471
|
+
'',
|
|
472
|
+
'当前教训文件:',
|
|
473
|
+
currentLessons,
|
|
474
|
+
'',
|
|
475
|
+
'要求:',
|
|
476
|
+
'1. 找两条「主题重叠」的教训(例如都讲「读码优先」、都讲「隔离」),不要选主题无关的两条。',
|
|
477
|
+
'2. titleA / titleB 是 `## ` 之后、标题的完整文本(含 category 前缀,逐字复制,不要改写;**不含** `## ` 前缀本身)。',
|
|
478
|
+
'3. merged 是合并后的综合教训:category 沿用其中一个、title 概括两者、suggestion 综合两条的核心建议、evidence 综合两条的证据要点。',
|
|
479
|
+
'4. 只返回裸 JSON(不要 markdown 围栏、不要其他文字),格式:',
|
|
480
|
+
'{"titleA":"<完整 ## 行1>","titleB":"<完整 ## 行2>","merged":{"category":"...","title":"...","suggestion":"...","evidence":["...","..."]}}',
|
|
481
|
+
].join('\n')
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/** 剥 ```json 围栏(LLM 可能加)。 */
|
|
485
|
+
function stripJsonFence(text: string): string {
|
|
486
|
+
const match = text.match(/^```(?:json)?\s*\n([\s\S]*?)\n```\s*$/)
|
|
487
|
+
return match ? match[1]! : text
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
/** Crossover 结果:两条教训的完整 ## 行 + 合并版。 */
|
|
491
|
+
export interface CrossoverResult {
|
|
492
|
+
titleA: string
|
|
493
|
+
titleB: string
|
|
494
|
+
merged: CrsiSignal
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
/** 解析 crossover 结果;非法 / 字段缺失 → null。 */
|
|
498
|
+
export function parseCrossoverResult(text: string): CrossoverResult | null {
|
|
499
|
+
try {
|
|
500
|
+
const obj = JSON.parse(stripJsonFence(text))
|
|
501
|
+
if (typeof obj.titleA !== 'string' || typeof obj.titleB !== 'string') return null
|
|
502
|
+
if (
|
|
503
|
+
!obj.merged ||
|
|
504
|
+
typeof obj.merged.category !== 'string' ||
|
|
505
|
+
typeof obj.merged.title !== 'string' ||
|
|
506
|
+
typeof obj.merged.suggestion !== 'string'
|
|
507
|
+
)
|
|
508
|
+
return null
|
|
509
|
+
const evidence = Array.isArray(obj.merged.evidence)
|
|
510
|
+
? obj.merged.evidence.filter((e: unknown) => typeof e === 'string')
|
|
511
|
+
: []
|
|
512
|
+
return {
|
|
513
|
+
titleA: obj.titleA,
|
|
514
|
+
titleB: obj.titleB,
|
|
515
|
+
merged: {
|
|
516
|
+
category: obj.merged.category,
|
|
517
|
+
title: obj.merged.title,
|
|
518
|
+
suggestion: obj.merged.suggestion,
|
|
519
|
+
evidence,
|
|
520
|
+
},
|
|
521
|
+
}
|
|
522
|
+
} catch {
|
|
523
|
+
return null
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
/** 从教训文件移除若干 `## ` 段(header 须是 `## ` 行完整文本)。preamble 与其余教训不动。 */
|
|
528
|
+
export function removeLessonSections(content: string, headers: string[]): string {
|
|
529
|
+
const lines = content.split('\n')
|
|
530
|
+
const out: string[] = []
|
|
531
|
+
let skipping = false
|
|
532
|
+
for (const line of lines) {
|
|
533
|
+
if (line.startsWith('## ')) {
|
|
534
|
+
skipping = headers.includes(line.trim())
|
|
535
|
+
if (skipping) continue
|
|
536
|
+
}
|
|
537
|
+
if (skipping) continue
|
|
538
|
+
out.push(line)
|
|
539
|
+
}
|
|
540
|
+
return out.join('\n')
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
/**
|
|
544
|
+
* Crossover:合并两条重叠教训 → 「删二增一」的教训文件变更候选。
|
|
545
|
+
* LLM 只生成(选对 + 合并版),guard 校验所选教训真实存在(fail-closed 防幻觉)。
|
|
546
|
+
*/
|
|
547
|
+
export async function produceCrossoverProposal(
|
|
548
|
+
llm: Llm,
|
|
549
|
+
currentLessons: string,
|
|
550
|
+
timestamp: string,
|
|
551
|
+
): Promise<{
|
|
552
|
+
description: string
|
|
553
|
+
filePath: string
|
|
554
|
+
newContent: string
|
|
555
|
+
originalContent: string
|
|
556
|
+
blastRadius: string[]
|
|
557
|
+
} | null> {
|
|
558
|
+
const response = await collectLlmText(llm, buildCrossoverPrompt(currentLessons))
|
|
559
|
+
if (!response) return null
|
|
560
|
+
|
|
561
|
+
const parsed = parseCrossoverResult(response)
|
|
562
|
+
if (!parsed) return null
|
|
563
|
+
|
|
564
|
+
const headerA = `## ${parsed.titleA}`
|
|
565
|
+
const headerB = `## ${parsed.titleB}`
|
|
566
|
+
if (parsed.titleA === parsed.titleB) return null
|
|
567
|
+
// 精确行匹配(与 removeLessonSections 同语义):子串 includes 会让「截断标题」漏过 guard、
|
|
568
|
+
// 却因 removeLessonSections 精确匹配删不掉 → 假「删二增一」实为「增一」。fail-closed 用精确匹配。
|
|
569
|
+
const lessonLines = currentLessons.split('\n').map((l) => l.trim())
|
|
570
|
+
if (!lessonLines.includes(headerA) || !lessonLines.includes(headerB)) return null
|
|
571
|
+
|
|
572
|
+
const withoutTwo = removeLessonSections(currentLessons, [headerA, headerB])
|
|
573
|
+
const mergedSection = buildLessonContent(parsed.merged, timestamp, 'CRSI producer (crossover)')
|
|
574
|
+
const newContent = `${withoutTwo.trimEnd()}\n\n${mergedSection}\n`
|
|
575
|
+
|
|
576
|
+
return {
|
|
577
|
+
description: `CRSI crossover: ${parsed.titleA} + ${parsed.titleB}`,
|
|
578
|
+
filePath: LESSONS_FILE,
|
|
579
|
+
newContent,
|
|
580
|
+
originalContent: currentLessons,
|
|
581
|
+
blastRadius: [LESSONS_FILE],
|
|
582
|
+
}
|
|
583
|
+
}
|
package/src/core/crsi-sandbox.ts
CHANGED
|
@@ -88,39 +88,77 @@ const TEST_TIMEOUT_MS = 120_000 // 2 minutes
|
|
|
88
88
|
const REPORT_DIR = join(homedir(), '.mipham', 'crsi-sandbox')
|
|
89
89
|
|
|
90
90
|
/**
|
|
91
|
-
*
|
|
91
|
+
* 自改进的「不可变基础」(immutable base)——按语义角色三类。
|
|
92
|
+
* 自改进循环可以改 skill/workflow/prompt/memory/教训/managed-rules,
|
|
93
|
+
* 但绝不能改以下三者,否则会削弱 grader 或安全边界:
|
|
94
|
+
* constitution —— 宪法/对齐:改掉 = 价值漂移优化掉安全边界
|
|
95
|
+
* evaluator —— 评估器/grader:改掉 = 改掉自己的评分标准(Goodhart 元劫持)
|
|
96
|
+
* selfImprovement —— 改进机制自身:改掉 = 递归改掉评估器/安全
|
|
92
97
|
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
98
|
+
* 注意:fail-closed——宁可多拦,不可漏拦。新增机制文件必须加进对应类别,
|
|
99
|
+
* 否则 eval harness 的 protection-completeness 契约会 fail。
|
|
100
|
+
*/
|
|
101
|
+
export const PROTECTED_ROLES = {
|
|
102
|
+
constitution: [
|
|
103
|
+
'apps/cli/src/core/alignment-vocabulary.json',
|
|
104
|
+
'apps/cli/src/core/constitution-loader.ts',
|
|
105
|
+
'apps/cli/src/core/constitution-seam.ts',
|
|
106
|
+
'apps/cli/src/vajra/constitution.ts',
|
|
107
|
+
],
|
|
108
|
+
evaluator: [
|
|
109
|
+
'apps/cli/test/',
|
|
110
|
+
'apps/cli/src/core/eval-harness.ts',
|
|
111
|
+
'apps/cli/src/core/behavior-tasks.ts',
|
|
112
|
+
'apps/cli/src/core/behavior-tasks.json',
|
|
113
|
+
'apps/cli/src/core/task-performance.ts',
|
|
114
|
+
'apps/cli/src/core/task-performance-tasks.json',
|
|
115
|
+
'apps/cli/src/core/improvement-track.ts',
|
|
116
|
+
'apps/cli/src/core/reward-fn.ts',
|
|
117
|
+
],
|
|
118
|
+
selfImprovement: [
|
|
119
|
+
'apps/cli/src/agent/effectiveness-tracker.ts',
|
|
120
|
+
'apps/cli/src/agent/recoverable-failure.ts',
|
|
121
|
+
'apps/cli/src/agent/crsi-provenance-bridge.ts',
|
|
122
|
+
'apps/cli/src/agent/experience-rules.ts',
|
|
123
|
+
'apps/cli/src/agent/agent-experience.ts',
|
|
124
|
+
'apps/cli/src/core/meta-rule-engine.ts',
|
|
125
|
+
'apps/cli/src/core/crsi-sandbox.ts',
|
|
126
|
+
'apps/cli/src/core/crsi-producer.ts',
|
|
127
|
+
'apps/cli/src/core/proposal-guard.ts',
|
|
128
|
+
'apps/cli/src/core/crsi-modify.ts',
|
|
129
|
+
'apps/cli/src/core/rule-engine.ts',
|
|
130
|
+
'apps/cli/src/core/red-team.ts',
|
|
131
|
+
'apps/cli/src/core/error-signature-db.ts',
|
|
132
|
+
'apps/cli/src/core/preflight-checker.ts',
|
|
133
|
+
'apps/cli/src/core/permission-rules.ts',
|
|
134
|
+
'apps/cli/src/core/rules-loader.ts',
|
|
135
|
+
],
|
|
136
|
+
} as const
|
|
137
|
+
|
|
138
|
+
/** 扁平化(向后兼容:isProtectedPath 仍用前缀匹配,行为不变)。 */
|
|
139
|
+
export const PROTECTED_PATHS: string[] = Object.values(PROTECTED_ROLES).flat()
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* 完整性金丝雀:这些「评估器 + 核心机制」文件必须全在保护域。
|
|
143
|
+
* eval harness 的 protection-completeness 契约逐条断言 isProtectedPath。
|
|
144
|
+
* 与 PROTECTED_ROLES 同文件(单一维护点)——新增机制文件须两处一起加。
|
|
100
145
|
*/
|
|
101
|
-
const
|
|
102
|
-
// 宪法(对齐)
|
|
103
|
-
'apps/cli/src/core/alignment-vocabulary.json',
|
|
104
|
-
'apps/cli/src/core/constitution-loader.ts',
|
|
105
|
-
'apps/cli/src/core/constitution-seam.ts',
|
|
106
|
-
'apps/cli/src/vajra/constitution.ts',
|
|
107
|
-
// eval harness
|
|
108
|
-
'apps/cli/test/',
|
|
146
|
+
export const PROTECTED_CRITICAL_FILES: string[] = [
|
|
109
147
|
'apps/cli/src/core/eval-harness.ts',
|
|
110
148
|
'apps/cli/src/core/behavior-tasks.ts',
|
|
111
149
|
'apps/cli/src/core/behavior-tasks.json',
|
|
112
|
-
|
|
113
|
-
'apps/cli/src/
|
|
114
|
-
'apps/cli/src/core/
|
|
150
|
+
'apps/cli/src/core/task-performance.ts',
|
|
151
|
+
'apps/cli/src/core/task-performance-tasks.json',
|
|
152
|
+
'apps/cli/src/core/improvement-track.ts',
|
|
153
|
+
'apps/cli/src/core/reward-fn.ts',
|
|
115
154
|
'apps/cli/src/core/crsi-sandbox.ts',
|
|
116
155
|
'apps/cli/src/core/crsi-producer.ts',
|
|
117
|
-
'apps/cli/src/core/proposal-guard.ts',
|
|
118
|
-
// 2026-08-27 review 补齐:运行时执行/评估器,改掉会削弱 grader 或安全边界
|
|
119
|
-
'apps/cli/src/core/crsi-modify.ts',
|
|
120
156
|
'apps/cli/src/core/rule-engine.ts',
|
|
121
157
|
'apps/cli/src/core/red-team.ts',
|
|
122
158
|
'apps/cli/src/core/error-signature-db.ts',
|
|
123
159
|
'apps/cli/src/core/preflight-checker.ts',
|
|
160
|
+
'apps/cli/src/agent/recoverable-failure.ts',
|
|
161
|
+
'apps/cli/src/agent/crsi-provenance-bridge.ts',
|
|
124
162
|
]
|
|
125
163
|
|
|
126
164
|
/** 是否命中只读边界。前缀匹配,目录条目以 `/` 结尾。 */
|
package/src/core/eval-harness.ts
CHANGED
|
@@ -19,7 +19,7 @@ import { ConstitutionLoader, DEFAULT_CONSTITUTION } from './constitution-loader'
|
|
|
19
19
|
import { ErrorSignatureDB } from './error-signature-db'
|
|
20
20
|
import { PreFlightChecker } from './preflight-checker'
|
|
21
21
|
import { RedTeam } from './red-team'
|
|
22
|
-
import { isProtectedPath, validateBlastRadius } from './crsi-sandbox'
|
|
22
|
+
import { isProtectedPath, validateBlastRadius, PROTECTED_CRITICAL_FILES } from './crsi-sandbox'
|
|
23
23
|
import { produceRuleProposal, MANAGED_RULES_FILE } from './crsi-producer'
|
|
24
24
|
import type { CrsiSignal } from './crsi-producer'
|
|
25
25
|
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
@@ -46,13 +46,17 @@ export interface EvalReport {
|
|
|
46
46
|
|
|
47
47
|
const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
|
|
48
48
|
|
|
49
|
-
/** 追加一次评估分数到 rewards
|
|
50
|
-
export function appendEvalScore(
|
|
49
|
+
/** 追加一次评估分数到 rewards 日志(按奖励函数名键控)。 */
|
|
50
|
+
export function appendEvalScore(
|
|
51
|
+
name: string,
|
|
52
|
+
report: { score: number; passed: number; total: number },
|
|
53
|
+
): void {
|
|
51
54
|
try {
|
|
52
55
|
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
53
56
|
appendFileSync(
|
|
54
57
|
SCORES_FILE,
|
|
55
58
|
JSON.stringify({
|
|
59
|
+
name,
|
|
56
60
|
timestamp: new Date().toISOString(),
|
|
57
61
|
score: report.score,
|
|
58
62
|
passed: report.passed,
|
|
@@ -65,14 +69,16 @@ export function appendEvalScore(report: EvalReport): void {
|
|
|
65
69
|
}
|
|
66
70
|
}
|
|
67
71
|
|
|
68
|
-
/**
|
|
69
|
-
export function getLastEvalScore(): number | null {
|
|
72
|
+
/** 读取某奖励函数最近一次分数(无记录时返回 null)。旧无 name 记录自然跳过。 */
|
|
73
|
+
export function getLastEvalScore(name: string): number | null {
|
|
70
74
|
try {
|
|
71
75
|
if (!existsSync(SCORES_FILE)) return null
|
|
72
76
|
const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
77
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
78
|
+
const rec = JSON.parse(lines[i]!) as { name?: string; score?: number }
|
|
79
|
+
if (rec.name === name && typeof rec.score === 'number') return rec.score
|
|
80
|
+
}
|
|
81
|
+
return null
|
|
76
82
|
} catch {
|
|
77
83
|
return null
|
|
78
84
|
}
|
|
@@ -167,6 +173,15 @@ export function runEval(): EvalReport {
|
|
|
167
173
|
results.push({ id, description: `受保护路径被拒: ${path}`, passed: isProtectedPath(path) })
|
|
168
174
|
}
|
|
169
175
|
|
|
176
|
+
// ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
|
|
177
|
+
const unprotected = PROTECTED_CRITICAL_FILES.filter((f) => !isProtectedPath(f))
|
|
178
|
+
results.push({
|
|
179
|
+
id: 'protection-completeness',
|
|
180
|
+
description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
|
|
181
|
+
passed: unprotected.length === 0,
|
|
182
|
+
...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
|
|
183
|
+
})
|
|
184
|
+
|
|
170
185
|
// ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
|
|
171
186
|
results.push({
|
|
172
187
|
id: 'blast-radius-gate',
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
// CRSI 改进轨:噪声自适应改进判定 + 台账 + pending verdict 闸。
|
|
2
|
+
// A1 不破:verdict / minEffect / 改进率全是确定性算术(均值/标准差/阈值/Wilson),无 LLM 裁判。
|
|
3
|
+
import { appendFileSync, readFileSync, existsSync, mkdirSync } from 'node:fs'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import { homedir } from 'node:os'
|
|
6
|
+
import type { SkillDeltaSample } from './task-performance'
|
|
7
|
+
|
|
8
|
+
export type ImprovementVerdict = 'improved' | 'regressed' | 'inconclusive'
|
|
9
|
+
|
|
10
|
+
export const MIN_EFFECT_FLOOR = 20
|
|
11
|
+
export const NOISE_K = 2
|
|
12
|
+
export const FALSE_POSITIVE_BASELINE = 0.05
|
|
13
|
+
|
|
14
|
+
export interface ImprovementReport {
|
|
15
|
+
skillName: string
|
|
16
|
+
changeSet: string[]
|
|
17
|
+
causal: boolean
|
|
18
|
+
baselineScores: number[]
|
|
19
|
+
postScores: number[]
|
|
20
|
+
deltaMean: number
|
|
21
|
+
noise: number
|
|
22
|
+
minEffect: number
|
|
23
|
+
verdict: ImprovementVerdict
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface ImprovementRecord extends ImprovementReport {
|
|
27
|
+
id: string
|
|
28
|
+
timestamp: string
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// ── 纯函数 ──
|
|
32
|
+
|
|
33
|
+
export function computeMinEffect(noise: number): number {
|
|
34
|
+
return Math.max(MIN_EFFECT_FLOOR, NOISE_K * noise)
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function classifyDelta(deltaMean: number, minEffect: number): ImprovementVerdict {
|
|
38
|
+
if (deltaMean <= -minEffect) return 'regressed'
|
|
39
|
+
if (deltaMean >= minEffect) return 'improved'
|
|
40
|
+
return 'inconclusive'
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function mean(xs: number[]): number {
|
|
44
|
+
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function stdDev(xs: number[]): number {
|
|
48
|
+
if (xs.length < 2) return 0
|
|
49
|
+
const m = mean(xs)
|
|
50
|
+
const variance = xs.reduce((a, b) => a + (b - m) ** 2, 0) / (xs.length - 1)
|
|
51
|
+
return Math.sqrt(variance)
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function buildImprovementReport(
|
|
55
|
+
sample: SkillDeltaSample,
|
|
56
|
+
changeSet: string[],
|
|
57
|
+
): ImprovementReport {
|
|
58
|
+
const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
|
|
59
|
+
const noise = stdDev(sample.baselineScores)
|
|
60
|
+
const minEffect = computeMinEffect(noise)
|
|
61
|
+
const verdict = classifyDelta(deltaMean, minEffect)
|
|
62
|
+
return {
|
|
63
|
+
skillName: sample.skillName,
|
|
64
|
+
changeSet,
|
|
65
|
+
causal: changeSet.length === 1,
|
|
66
|
+
baselineScores: sample.baselineScores,
|
|
67
|
+
postScores: sample.postScores,
|
|
68
|
+
deltaMean,
|
|
69
|
+
noise,
|
|
70
|
+
minEffect,
|
|
71
|
+
verdict,
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Wilson score 区间(z 默认 1.96 = 95%)。n=0 → {0,0}。 */
|
|
76
|
+
export function wilsonInterval(
|
|
77
|
+
improved: number,
|
|
78
|
+
total: number,
|
|
79
|
+
z = 1.96,
|
|
80
|
+
): { lo: number; hi: number } {
|
|
81
|
+
if (total === 0) return { lo: 0, hi: 0 }
|
|
82
|
+
const p = improved / total
|
|
83
|
+
const n = total
|
|
84
|
+
const denom = 1 + (z * z) / n
|
|
85
|
+
const center = (p + (z * z) / (2 * n)) / denom
|
|
86
|
+
const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / denom
|
|
87
|
+
return { lo: center - half, hi: center + half }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function improvementRate(records: ImprovementRecord[]): {
|
|
91
|
+
total: number
|
|
92
|
+
improved: number
|
|
93
|
+
rate: number
|
|
94
|
+
lo: number
|
|
95
|
+
hi: number
|
|
96
|
+
} {
|
|
97
|
+
const total = records.length
|
|
98
|
+
const improved = records.filter((r) => r.verdict === 'improved').length
|
|
99
|
+
const { lo, hi } = wilsonInterval(improved, total)
|
|
100
|
+
return { total, improved, rate: total === 0 ? 0 : improved / total, lo, hi }
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** 循环有效性:改进率 Wilson 95% CI 下界 > 假阳性基线(默认 5%)。 */
|
|
104
|
+
export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
|
|
105
|
+
const { lo } = improvementRate(records)
|
|
106
|
+
return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// ── 台账 ──
|
|
110
|
+
|
|
111
|
+
export function improvementPath(): string {
|
|
112
|
+
return join(homedir(), '.mipham', 'crsi', 'improvements.jsonl')
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export function appendImprovement(record: ImprovementRecord): void {
|
|
116
|
+
const file = improvementPath()
|
|
117
|
+
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
118
|
+
appendFileSync(file, JSON.stringify(record) + '\n', 'utf-8')
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export function readImprovements(): ImprovementRecord[] {
|
|
122
|
+
const file = improvementPath()
|
|
123
|
+
if (!existsSync(file)) return []
|
|
124
|
+
return readFileSync(file, 'utf-8')
|
|
125
|
+
.split('\n')
|
|
126
|
+
.filter((line) => line.trim() !== '')
|
|
127
|
+
.map((line) => JSON.parse(line) as ImprovementRecord)
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// ── pending verdict 闸(倒退才拦) ──
|
|
131
|
+
|
|
132
|
+
let pendingVerdict: ImprovementVerdict | null = null
|
|
133
|
+
|
|
134
|
+
export function setPendingVerdict(v: ImprovementVerdict | null): void {
|
|
135
|
+
pendingVerdict = v
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export function getPendingVerdict(): ImprovementVerdict | null {
|
|
139
|
+
return pendingVerdict
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
export function shouldBlockApproval(v: ImprovementVerdict): boolean {
|
|
143
|
+
return v === 'regressed'
|
|
144
|
+
}
|
package/src/core/instructions.ts
CHANGED
|
@@ -263,6 +263,18 @@ its live CRSI / SIS / constitution state. Report the numbers you read
|
|
|
263
263
|
from it as live counts; if it shows a subsystem as 未初始化 (uninitialized),
|
|
264
264
|
say so explicitly instead of claiming it exists.`)
|
|
265
265
|
|
|
266
|
+
// CRSI 先读代码铁律 — 回答代码问题前必须先读实际代码,勿凭记忆/命名/静态清单下结论
|
|
267
|
+
parts.push(`## Read-Code-First Rule
|
|
268
|
+
|
|
269
|
+
Before answering ANY question about this codebase — whether a file,
|
|
270
|
+
function, feature, or capability exists, how it works, or whether
|
|
271
|
+
something is missing — you MUST first read the actual code with the
|
|
272
|
+
Read, Grep, Glob, or graft tools. Do not infer or assert from memory,
|
|
273
|
+
naming conventions, or static tool lists. If you have not read the code,
|
|
274
|
+
say so and read it first, rather than answering hastily and retracting
|
|
275
|
+
afterwards. This applies to every code question, not only research or
|
|
276
|
+
borrow-analysis tasks.`)
|
|
277
|
+
|
|
266
278
|
// CRSI 教训召回 — 把 crsi-lessons.md 的教训精华注入,让模型「写后召回」而非只写不读
|
|
267
279
|
const lessonsBlock = buildCrsiLessonsBlock(this.crsiLessonSummaries)
|
|
268
280
|
if (lessonsBlock) parts.push(lessonsBlock)
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// CRSI RewardFn 接口——reward function = policy→feedback 抽象。
|
|
2
|
+
// 统一「给一个 policy 打分」:机制哨兵(runEval)与任务表现(runTaskPerformance)
|
|
3
|
+
// 都 conform 成 RewardFn,自改进环的 verify 阶段可对任意奖励源比分数、判退化。
|
|
4
|
+
import type { Llm } from '../providers/llm'
|
|
5
|
+
import { runEval } from './eval-harness'
|
|
6
|
+
import { runTaskPerformance } from './task-performance'
|
|
7
|
+
|
|
8
|
+
/** 奖励函数统一输出的「分数」形状——所有 RewardFn 的 evaluate 都产出它。 */
|
|
9
|
+
export interface ScoreReport {
|
|
10
|
+
total: number
|
|
11
|
+
passed: number
|
|
12
|
+
score: number // 0-100
|
|
13
|
+
failures: string[]
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/** 奖励函数(reward function = policy→feedback):一个具名、可替换的打分器。 */
|
|
17
|
+
export interface RewardFn {
|
|
18
|
+
name: string
|
|
19
|
+
description: string
|
|
20
|
+
evaluate(): Promise<ScoreReport> | ScoreReport
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** 机制哨兵:冻结契约评当前仓库机制代码(无 LLM,同步)。 */
|
|
24
|
+
export function mechanismSentinel(): RewardFn {
|
|
25
|
+
return {
|
|
26
|
+
name: 'mechanism-sentinel',
|
|
27
|
+
description: '冻结契约评当前仓库机制代码(无 LLM)',
|
|
28
|
+
evaluate: () => runEval(),
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** 任务表现:LLM 生成代码 + 冻结测试判定(有 LLM,异步)。 */
|
|
33
|
+
export function taskPerformanceRewardFn(llm: Llm): RewardFn {
|
|
34
|
+
return {
|
|
35
|
+
name: 'task-performance',
|
|
36
|
+
description: 'LLM 生成 + 冻结测试评 skill/通用任务',
|
|
37
|
+
evaluate: () => runTaskPerformance(llm),
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** 可枚举的奖励函数注册表。llm 缺省时只含无 LLM 的机制哨兵。 */
|
|
42
|
+
export function listRewardFns(llm?: Llm): RewardFn[] {
|
|
43
|
+
return [mechanismSentinel(), ...(llm ? [taskPerformanceRewardFn(llm)] : [])]
|
|
44
|
+
}
|