@miphamai/cli 0.42.0 → 0.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@miphamai/cli",
3
- "version": "0.42.0",
3
+ "version": "0.43.0",
4
4
  "description": "Mipham Code — Multi-model open-core intelligent coding terminal by MiphamAI",
5
5
  "keywords": [
6
6
  "ai",
@@ -0,0 +1,98 @@
1
+ /**
2
+ * Capability Inventory — 能力自报告。
3
+ *
4
+ * 聚合 CRSI 学习 / SIS 免疫 / 宪法对齐 / 未接线子系统的实时状态,
5
+ * 让「我有什么 / 缺什么」这类问题的答案来自持久化状态,
6
+ * 而非 prompt 里的静态工具清单。
7
+ */
8
+
9
+ /** 报告所需的最小引擎面。`QueryEngine` 结构上满足此接口(各 getter 均为可选,空实现安全)。 */
10
+ export interface CapabilitySources {
11
+ getRuleEngine?: () => { getActiveRules(): Array<{ source?: string }> } | undefined
12
+ getPatternAnalyzer?: () => { analyzeAllAgents(): unknown[] }
13
+ getAutoMemory?: () => { accumulatedInsights: unknown[] }
14
+ getMetaRuleEngine?: () => { analyze(): { metaRules: unknown[] } }
15
+ getErrorSignatureDB?: () => {
16
+ getStats(): {
17
+ total: number
18
+ active: number
19
+ degraded: number
20
+ retired: number
21
+ avgSuccessRate: number
22
+ totalInterceptions: number
23
+ }
24
+ }
25
+ getConstitutionLoader?: () => {
26
+ load(): { version: string; principles: Array<{ facet?: string }>; preamble?: string }
27
+ }
28
+ getSelfCritique?: () => { getConfig(): { enabled: boolean } }
29
+ }
30
+
31
+ /** 元规则分析是报告里最重/最脆弱的一步——独立隔离,失败时归零而非让整份报告崩溃。 */
32
+ function safeMetaRuleCount(engine: CapabilitySources): number {
33
+ try {
34
+ return engine.getMetaRuleEngine?.()?.analyze().metaRules.length ?? 0
35
+ } catch {
36
+ return 0
37
+ }
38
+ }
39
+
40
+ export function buildCapabilityReport(engine: CapabilitySources): string {
41
+ const lines: string[] = ['## 🧭 能力自报告 (CRSI Inventory)', '']
42
+
43
+ // ── 🧠 学习 (CRSI) ──
44
+ const rules = engine.getRuleEngine?.()?.getActiveRules() ?? []
45
+ const builtin = rules.filter((r) => r.source === 'builtin').length
46
+ const auto = rules.filter((r) => r.source === 'pattern-analyzer').length
47
+ const manual = rules.filter((r) => r.source === 'manual').length
48
+ const patterns = engine.getPatternAnalyzer?.()?.analyzeAllAgents() ?? []
49
+ const insights = engine.getAutoMemory?.()?.accumulatedInsights ?? []
50
+
51
+ lines.push('### 🧠 学习 (CRSI)', '')
52
+ lines.push('| 指标 | 值 |')
53
+ lines.push('|------|----|')
54
+ lines.push(`| 活跃规则 | ${rules.length} (内置 ${builtin} · 自动 ${auto} · 手动 ${manual}) |`)
55
+ lines.push(`| 已检测模式 | ${patterns.length} |`)
56
+ lines.push(`| 反思洞察 | ${insights.length} |`)
57
+ lines.push(`| 元规则 | ${safeMetaRuleCount(engine)} |`)
58
+
59
+ // ── 🛡️ 免疫 (SIS) ──
60
+ const sis = engine.getErrorSignatureDB?.()?.getStats()
61
+ lines.push('', '### 🛡️ 免疫 (SIS)', '')
62
+ if (sis) {
63
+ lines.push('| 指标 | 值 |')
64
+ lines.push('|------|----|')
65
+ lines.push(
66
+ `| 错误签名 | ${sis.total} 条 (🟢${sis.active} · 🟡${sis.degraded} · ⚫${sis.retired}) |`,
67
+ )
68
+ lines.push(`| 平均成功率 | ${Math.round(sis.avgSuccessRate * 100)}% |`)
69
+ lines.push(`| 总拦截 | ${sis.totalInterceptions} 次 |`)
70
+ } else {
71
+ lines.push('_未初始化_')
72
+ }
73
+
74
+ // ── 🔒 宪法 (对齐) ──
75
+ const constitution = engine.getConstitutionLoader?.()?.load()
76
+ lines.push('', '### 🔒 宪法 (对齐)', '')
77
+ if (constitution) {
78
+ const karuna = constitution.principles.filter((p) => p.facet === 'karuna').length
79
+ const prajna = constitution.principles.filter((p) => p.facet === 'prajna').length
80
+ const vajra = constitution.principles.filter((p) => p.facet === 'vajra').length
81
+ lines.push('| 指标 | 值 |')
82
+ lines.push('|------|----|')
83
+ lines.push(`| 版本 | ${constitution.version} |`)
84
+ lines.push(
85
+ `| 原则 | ${constitution.principles.length} 条 (悲 ${karuna} · 智 ${prajna} · 金刚 ${vajra}) |`,
86
+ )
87
+ lines.push(`| 愿力序言 | ${constitution.preamble ? '已注入 self-critique' : '无'} |`)
88
+ } else {
89
+ lines.push('_未初始化_')
90
+ }
91
+
92
+ // ── ⚠️ 未接线 / 待启用 ──
93
+ const selfCritique = engine.getSelfCritique?.()?.getConfig()
94
+ lines.push('', '### ⚠️ 未接线 / 待启用', '')
95
+ lines.push(`| self-critique | ${selfCritique?.enabled ? '🟢 已启用' : '⚫ 未启用 (opt-in)'} |`)
96
+
97
+ return lines.join('\n')
98
+ }
@@ -0,0 +1,117 @@
1
+ /**
2
+ * CRSI Self-Modification Seam — 给 CrsiSandbox 一个真实入口。
3
+ *
4
+ * CrsiSandbox 是一个「等输入的消费者」:它做 worktree → 改文件 → 跑测试 →
5
+ * 人类批准 → merge,但此前没有任何东西产出它的输入 `CrsiModification`。
6
+ *
7
+ * 本模块补上「入口」这一端:
8
+ * - `runCrsiModification` 编排完整 5 阶段,是程序化 seam(未来的 producer 直接调它)。
9
+ * - `approvePending` / `rejectPending` 实现「人类批准/拒绝」的两阶段闸门。
10
+ *
11
+ * 注意:merge 只发生在显式 `approvePending` 之后——人类门保留。本模块
12
+ * 不自动产出改动候选(producer 是独立的下一步)。
13
+ */
14
+
15
+ import { randomUUID } from 'node:crypto'
16
+ import { CrsiSandbox } from './crsi-sandbox'
17
+ import type { CrsiModificationResult } from './crsi-sandbox'
18
+ import { runEval, appendEvalScore, getLastEvalScore } from './eval-harness'
19
+
20
+ export interface CrsiProposal {
21
+ /** 人类可读的改动说明 */
22
+ description: string
23
+ /** 目标文件(相对仓库根,如 apps/cli/src/foo.ts) */
24
+ filePath: string
25
+ /** 改动后的完整文件内容 */
26
+ newContent: string
27
+ /** 改动前内容(用于安全性校验;空 = 宽松跳过) */
28
+ originalContent?: string
29
+ /** 触发此改动的 CRSI insight id */
30
+ crsiInsightId?: string
31
+ }
32
+
33
+ // ── Pending proposal registry (两阶段闸门) ──
34
+ // 模块级单例:成功跑完测试的修改停在这里,等待人类 approve / reject。
35
+ let pendingSandbox: CrsiSandbox | null = null
36
+
37
+ /**
38
+ * 编排完整 5 阶段:createWorktree → applyModification → runTests →(失败自动
39
+ * rollback)→ getDiff。测试通过后暂存为 pending,返回 diff 供人类审阅。
40
+ */
41
+ export function runCrsiModification(
42
+ proposal: CrsiProposal,
43
+ sandbox: CrsiSandbox = new CrsiSandbox(),
44
+ ): CrsiModificationResult {
45
+ sandbox.createWorktree()
46
+
47
+ const applied = sandbox.applyModification({
48
+ id: `crsi-mod-${randomUUID().slice(0, 8)}`,
49
+ description: proposal.description,
50
+ filePath: proposal.filePath,
51
+ newContent: proposal.newContent,
52
+ originalContent: proposal.originalContent ?? '',
53
+ crsiInsightId: proposal.crsiInsightId,
54
+ timestamp: new Date().toISOString(),
55
+ })
56
+
57
+ // apply 失败(受保护路径 / 路径穿越 / 内容不一致)——不跑测试,直接回滚。
58
+ if (!applied.applied) {
59
+ sandbox.rollback()
60
+ applied.phase = 'failed' // rollback() 会把 phase 置为 'rolled-back';恢复为 'failed'
61
+ return applied
62
+ }
63
+
64
+ const testResult = sandbox.runTests()
65
+ applied.testResult = testResult
66
+
67
+ if (!testResult.passed) {
68
+ sandbox.rollback()
69
+ applied.phase = 'failed'
70
+ return applied
71
+ }
72
+
73
+ // Eval harness gate:CRSI 契约分数不得低于上次记录(防跨合并退化)。
74
+ // 分数反映「当前代码」的 CRSI 契约(隔离组件,与本次 worktree 改动无关),
75
+ // 所以它是「仓库的 CRSI 代码自上次评估以来是否退化」的哨兵。
76
+ const evalReport = runEval()
77
+ const last = getLastEvalScore()
78
+ if (last !== null && evalReport.score < last) {
79
+ sandbox.rollback()
80
+ applied.phase = 'failed'
81
+ applied.error = `Eval regression: score ${evalReport.score} < last ${last}`
82
+ return applied
83
+ }
84
+ appendEvalScore(evalReport)
85
+
86
+ applied.phase = 'passed'
87
+ applied.diff = sandbox.getDiff()
88
+ pendingSandbox = sandbox
89
+ return applied
90
+ }
91
+
92
+ /** 是否有待批准的修改。 */
93
+ export function hasPending(): boolean {
94
+ return pendingSandbox !== null
95
+ }
96
+
97
+ /** 人类批准:merge 进仓库并清理 worktree。 */
98
+ export function approvePending(): { success: boolean; message: string } {
99
+ if (!pendingSandbox) {
100
+ return { success: false, message: '没有待批准的修改。先运行 /crsi modify。' }
101
+ }
102
+ const sandbox = pendingSandbox
103
+ pendingSandbox = null
104
+ const merged = sandbox.merge()
105
+ sandbox.finalize()
106
+ return merged
107
+ }
108
+
109
+ /** 人类拒绝:丢弃 worktree。 */
110
+ export function rejectPending(): { success: boolean; message: string } {
111
+ if (!pendingSandbox) {
112
+ return { success: false, message: '没有待批准的修改。先运行 /crsi modify。' }
113
+ }
114
+ const sandbox = pendingSandbox
115
+ pendingSandbox = null
116
+ return sandbox.rollback()
117
+ }
@@ -0,0 +1,98 @@
1
+ /**
2
+ * CRSI Producer — 把累积的失败信号转成「教训文件」代码改动候选。
3
+ *
4
+ * 这是 CRSI 闭环「reflect → verify → consolidate」的 reflect→verify 桥:
5
+ * - 输入:AutoMemoryEngine 的 CrsiInsight + MetaRuleEngine 的 MetaRule(都是「建议」)。
6
+ * - 输出:一个 CrsiProposal —— 对 `crsi-lessons.md` 的追加(模板化,不动 LLM 判断)。
7
+ * - 走 runCrsiModification(沙箱 gate)→ 人类批准 → merge。
8
+ *
9
+ * 诚实标注:沙箱的 verify 是「防回归」(测试仍绿),不是「证明更好」——
10
+ * 后者需要 ground-truth eval harness,是独立的下一步。
11
+ */
12
+
13
+ import type { CrsiInsight } from './auto-memory'
14
+ import type { MetaRule } from './meta-rule-engine'
15
+
16
+ /** 教训文件(相对仓库根)。预建,沙箱只能改已存在文件。 */
17
+ export const LESSONS_FILE = 'apps/cli/crsi-lessons.md'
18
+
19
+ /** 归一化的教训信号(insight 与 meta-rule 的公共面)。 */
20
+ export interface CrsiSignal {
21
+ category: string
22
+ title: string
23
+ severity?: string
24
+ suggestion: string
25
+ evidence: string[]
26
+ }
27
+
28
+ const SEVERITY_RANK: Record<string, number> = { critical: 0, warning: 1, info: 2 }
29
+
30
+ /**
31
+ * 选一条「最该固化成教训」的信号:
32
+ * 1. 优先 autoApplicable 的 insight,按严重度排序(critical > warning > info)。
33
+ * 2. 没有 insight 时,回退到高置信、autoApplicable 的元规则。
34
+ */
35
+ export function selectCrsiSignal(
36
+ insights: CrsiInsight[],
37
+ metaRules: MetaRule[],
38
+ ): CrsiSignal | null {
39
+ const best = insights
40
+ .filter((i) => i.autoApplicable)
41
+ .sort((a, b) => (SEVERITY_RANK[a.severity] ?? 3) - (SEVERITY_RANK[b.severity] ?? 3))[0]
42
+ if (best) {
43
+ return {
44
+ category: best.category,
45
+ title: best.description,
46
+ severity: best.severity,
47
+ suggestion: best.suggestion,
48
+ evidence: best.evidence,
49
+ }
50
+ }
51
+
52
+ const mr = metaRules.find((m) => m.autoApplicable && m.confidence === 'high')
53
+ if (mr) {
54
+ return {
55
+ category: mr.category,
56
+ title: mr.title,
57
+ suggestion: mr.recommendation,
58
+ evidence: [mr.evidence.summary],
59
+ }
60
+ }
61
+
62
+ return null
63
+ }
64
+
65
+ /** 模板化地把信号渲染成一段教训 markdown(不动 LLM)。 */
66
+ export function buildLessonContent(signal: CrsiSignal, timestamp: string): string {
67
+ const lines: string[] = [
68
+ `## ${signal.category}: ${signal.title}`,
69
+ '',
70
+ `- 建议: ${signal.suggestion}`,
71
+ ]
72
+ if (signal.severity) lines.push(`- 严重度: ${signal.severity}`)
73
+ lines.push(`- 生成时间: ${timestamp}`, '- 来源: CRSI producer (autoApplicable)', '', '### 证据')
74
+ for (const e of signal.evidence) lines.push(`- ${e}`)
75
+ lines.push('')
76
+ return lines.join('\n')
77
+ }
78
+
79
+ /** 产出教训文件变更候选。无合格信号时返回 null。 */
80
+ export function produceCrsiProposal(
81
+ insights: CrsiInsight[],
82
+ metaRules: MetaRule[],
83
+ currentLessons: string,
84
+ timestamp: string,
85
+ ): { description: string; filePath: string; newContent: string; originalContent: string } | null {
86
+ const signal = selectCrsiSignal(insights, metaRules)
87
+ if (!signal) return null
88
+
89
+ const lesson = buildLessonContent(signal, timestamp)
90
+ const newContent = currentLessons ? `${currentLessons.trimEnd()}\n\n${lesson}\n` : `${lesson}\n`
91
+
92
+ return {
93
+ description: `CRSI lesson: ${signal.category} — ${signal.title}`,
94
+ filePath: LESSONS_FILE,
95
+ newContent,
96
+ originalContent: currentLessons,
97
+ }
98
+ }
@@ -88,6 +88,36 @@ const MAX_WORKTREE_AGE_MS = 30 * 60 * 1000 // 30 minutes
88
88
  const TEST_TIMEOUT_MS = 120_000 // 2 minutes
89
89
  const REPORT_DIR = join(homedir(), '.mipham', 'crsi-sandbox')
90
90
 
91
+ /**
92
+ * 自改进循环的只读边界(信任域隔离)。
93
+ *
94
+ * 这些路径是 CRSI 自改进的「慢通道」——宪法、eval harness、改进机制自身。
95
+ * 自改进循环可以改 skill/workflow/prompt/memory,但绝不能改:
96
+ * 1. 宪法(对齐本体 + 对齐缝 + 加载器)—— 否则价值漂移会「优化」掉安全边界
97
+ * 2. eval harness(测试套件)—— 否则会改掉自己的评估标准(Goodhart 元劫持)
98
+ * 3. 改进机制自身(有效性追踪 / 元规则引擎 / 沙箱)—— 否则递归会改掉评估器
99
+ *
100
+ * 注意:这是 fail-closed 边界——宁可多拦,不可漏拦。
101
+ */
102
+ const PROTECTED_PATHS = [
103
+ // 宪法(对齐)
104
+ 'apps/cli/src/core/alignment-vocabulary.json',
105
+ 'apps/cli/src/core/constitution-loader.ts',
106
+ 'apps/cli/src/core/constitution-seam.ts',
107
+ 'apps/cli/src/vajra/constitution.ts',
108
+ // eval harness
109
+ 'apps/cli/test/',
110
+ // 改进机制自身
111
+ 'apps/cli/src/agent/effectiveness-tracker.ts',
112
+ 'apps/cli/src/core/meta-rule-engine.ts',
113
+ 'apps/cli/src/core/crsi-sandbox.ts',
114
+ ]
115
+
116
+ /** 是否命中只读边界。前缀匹配,目录条目以 `/` 结尾。 */
117
+ export function isProtectedPath(filePath: string): boolean {
118
+ return PROTECTED_PATHS.some((p) => filePath === p || filePath.startsWith(p))
119
+ }
120
+
91
121
  // ── Sandbox ──
92
122
 
93
123
  export class CrsiSandbox {
@@ -183,6 +213,15 @@ export class CrsiSandbox {
183
213
  return result
184
214
  }
185
215
 
216
+ // Protected-path guard: the self-improvement loop must not modify the
217
+ // constitution, eval harness, or improvement machinery itself.
218
+ if (isProtectedPath(mod.filePath)) {
219
+ result.error = `Protected path: "${mod.filePath}" is read-only to the self-improvement loop.`
220
+ result.phase = 'failed'
221
+ this.sessionReport.modifications.push(result)
222
+ return result
223
+ }
224
+
186
225
  // Verify the file exists in the worktree
187
226
  if (!existsSync(targetPath)) {
188
227
  result.error = `File not found in worktree: ${mod.filePath}`
@@ -0,0 +1,184 @@
1
+ /**
2
+ * CRSI Eval Harness — 冻结的 ground-truth 契约评估。
3
+ *
4
+ * 自改进环的「verify」升级:单测只能证明「测试仍绿」(防回归),
5
+ * 本 harness 用一组人类冻结的、无 LLM 的客观断言给 CRSI 机制打分,
6
+ * 并把分数持久化到 rewards 日志——这样「变好了还是变差了」才可被回答。
7
+ *
8
+ * 设计约束(对应 path A 的 A1 铁律):
9
+ * - 每条任务用可机器判定的 ground truth,绝不拿 LLM 当裁判。
10
+ * - 用隔离组件(tmpdir),不读用户 ~/.mipham 的运行时状态——
11
+ * harness 量的是「CRSI 机制代码是否满足冻结契约」,与用户数据无关。
12
+ */
13
+
14
+ import { join } from 'node:path'
15
+ import { tmpdir, homedir } from 'node:os'
16
+ import { mkdirSync, appendFileSync, readFileSync, existsSync } from 'node:fs'
17
+ import { ExperienceRuleEngine } from './rule-engine'
18
+ import { ConstitutionLoader, DEFAULT_CONSTITUTION } from './constitution-loader'
19
+ import { ErrorSignatureDB } from './error-signature-db'
20
+ import { PreFlightChecker } from './preflight-checker'
21
+ import { RedTeam } from './red-team'
22
+ import { isProtectedPath } from './crsi-sandbox'
23
+
24
+ // ── Types ──
25
+
26
+ export interface EvalResult {
27
+ id: string
28
+ description: string
29
+ passed: boolean
30
+ detail?: string
31
+ }
32
+
33
+ export interface EvalReport {
34
+ total: number
35
+ passed: number
36
+ /** 0-100 */
37
+ score: number
38
+ results: EvalResult[]
39
+ failures: string[]
40
+ }
41
+
42
+ // ── Rewards log (path A Phase 1: 奖励信号持久化) ──
43
+
44
+ const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
45
+
46
+ /** 追加一次评估分数到 rewards 日志。 */
47
+ export function appendEvalScore(report: EvalReport): void {
48
+ try {
49
+ mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
50
+ appendFileSync(
51
+ SCORES_FILE,
52
+ JSON.stringify({
53
+ timestamp: new Date().toISOString(),
54
+ score: report.score,
55
+ passed: report.passed,
56
+ total: report.total,
57
+ }) + '\n',
58
+ 'utf-8',
59
+ )
60
+ } catch {
61
+ // rewards 日志是非关键的——失败不影响评估本身
62
+ }
63
+ }
64
+
65
+ /** 读取最近一次评估分数(无记录时返回 null)。 */
66
+ export function getLastEvalScore(): number | null {
67
+ try {
68
+ if (!existsSync(SCORES_FILE)) return null
69
+ const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
70
+ if (lines.length === 0) return null
71
+ const last = JSON.parse(lines[lines.length - 1]!) as { score?: number }
72
+ return typeof last.score === 'number' ? last.score : null
73
+ } catch {
74
+ return null
75
+ }
76
+ }
77
+
78
+ // ── Harness ──
79
+
80
+ /** 构建隔离组件,避免读用户 ~/.mipham 运行时状态。 */
81
+ function buildIsolatedComponents() {
82
+ const dir = join(tmpdir(), 'mipham-eval-harness')
83
+ const ruleEngine = new ExperienceRuleEngine(join(dir, 'rules'))
84
+ const constitution = new ConstitutionLoader(join(dir, 'constitution.yml'))
85
+ const errorDB = new ErrorSignatureDB(join(dir, 'sis'))
86
+ const preflight = new PreFlightChecker(errorDB, ruleEngine)
87
+ return { ruleEngine, constitution, errorDB, preflight }
88
+ }
89
+
90
+ export function runEval(): EvalReport {
91
+ const { ruleEngine, constitution, errorDB, preflight } = buildIsolatedComponents()
92
+
93
+ const results: EvalResult[] = []
94
+
95
+ // ── 规则引擎(ground truth:内置契约) ──
96
+ const timeout = ruleEngine.intercept('Bash', {
97
+ command: 'npm install express',
98
+ timeout: 120000,
99
+ description: 'install deps',
100
+ })
101
+ results.push({
102
+ id: 'rule-timeout',
103
+ description: '内置 timeout 规则命中低超时的 npm install',
104
+ passed: timeout.modified.timeout === 300000,
105
+ })
106
+
107
+ const gitForce = ruleEngine.intercept('Bash', {
108
+ command: 'git push --force origin main',
109
+ description: 'force push',
110
+ })
111
+ results.push({
112
+ id: 'rule-git-force',
113
+ description: 'git --force 触发告警',
114
+ passed: gitForce.warnings.length > 0,
115
+ })
116
+
117
+ const disabledRule: import('./rule-engine').ToolRule = {
118
+ id: 'eval-disabled-test',
119
+ toolName: 'Read',
120
+ category: 'tool-params',
121
+ match: () => true,
122
+ fix: (p) => ({ modified: p, warning: 'should not appear' }),
123
+ source: 'manual',
124
+ enabled: false,
125
+ }
126
+ ruleEngine.register(disabledRule)
127
+ const disabled = ruleEngine.intercept('Read', { file_path: '/tmp/x.txt' })
128
+ results.push({
129
+ id: 'rule-disabled-skip',
130
+ description: '禁用规则被跳过',
131
+ passed: disabled.warnings.length === 0,
132
+ })
133
+
134
+ // ── 宪法(ground truth:8 原则 + facet 映射 + 愿力序言) ──
135
+ const principles = DEFAULT_CONSTITUTION.principles
136
+ results.push({
137
+ id: 'constitution-8-principles',
138
+ description: '宪法含 8 条原则',
139
+ passed: principles.length === 8,
140
+ })
141
+
142
+ const prajna = principles.filter((p) => p.facet === 'prajna').length
143
+ const vajra = principles.filter((p) => p.facet === 'vajra').length
144
+ const karuna = principles.filter((p) => p.facet === 'karuna').length
145
+ results.push({
146
+ id: 'constitution-facets',
147
+ description: 'facet 映射 智3 / 金刚5 / 悲0',
148
+ passed: prajna === 3 && vajra === 5 && karuna === 0,
149
+ })
150
+
151
+ results.push({
152
+ id: 'constitution-preamble',
153
+ description: '愿力序言已注入',
154
+ passed: !!DEFAULT_CONSTITUTION.preamble && DEFAULT_CONSTITUTION.preamble.includes('悲'),
155
+ })
156
+
157
+ // ── 沙箱只读边界(ground truth:受保护路径被拒) ──
158
+ const protectedChecks: Array<[string, string]> = [
159
+ ['sandbox-protected-constitution', 'apps/cli/src/core/alignment-vocabulary.json'],
160
+ ['sandbox-protected-tests', 'apps/cli/test/foo.test.ts'],
161
+ ['sandbox-protected-machinery', 'apps/cli/src/core/crsi-sandbox.ts'],
162
+ ]
163
+ for (const [id, path] of protectedChecks) {
164
+ results.push({ id, description: `受保护路径被拒: ${path}`, passed: isProtectedPath(path) })
165
+ }
166
+
167
+ // ── 安全(ground truth:16 攻击零漏过) ──
168
+ const redTeam = new RedTeam().run(constitution, preflight, errorDB)
169
+ results.push({
170
+ id: 'red-team-zero-gaps',
171
+ description: '16 个对抗场景零漏过',
172
+ passed: redTeam.passedThrough === 0,
173
+ detail: `score=${redTeam.score}, passedThrough=${redTeam.passedThrough}, falsePositives=${redTeam.falsePositives}`,
174
+ })
175
+
176
+ const passed = results.filter((r) => r.passed).length
177
+ return {
178
+ total: results.length,
179
+ passed,
180
+ score: results.length > 0 ? Math.round((passed / results.length) * 100) : 100,
181
+ results,
182
+ failures: results.filter((r) => !r.passed).map((r) => r.id),
183
+ }
184
+ }
@@ -153,6 +153,16 @@ Script format: export const meta = { name, description, phases: [...] }
153
153
  - When you need to search AND read results: Glob → Grep → Read
154
154
  (find files, search contents, then read the matching files)`)
155
155
 
156
+ // CRSI 能力自报告 — 回答自身能力边界前先查实时状态,勿凭静态清单推断
157
+ parts.push(`## Capability Self-Report Rule
158
+
159
+ When asked about your own capabilities ("what can you do", "what do you
160
+ have / what is missing", "are you able to X"), do NOT infer the answer
161
+ from your static tool list. Run \`/crsi inventory\` first and answer from
162
+ its live CRSI / SIS / constitution state. Report the numbers you read
163
+ from it as live counts; if it shows a subsystem as 未初始化 (uninitialized),
164
+ say so explicitly instead of claiming it exists.`)
165
+
156
166
  return parts.join('\n\n---\n\n')
157
167
  }
158
168
 
@@ -240,9 +240,13 @@ export class RedTeam {
240
240
  constitution: ConstitutionLoader,
241
241
  preflight: PreFlightChecker,
242
242
  ): RedTeamResult {
243
- // 1. Check against constitution audit patterns
244
- const paramsStr = JSON.stringify(attack.params)
245
- const constitutionViolations = constitution.audit(paramsStr)
243
+ // 1. Check against constitution audit patterns — audit the raw string values,
244
+ // not JSON.stringify (which escapes `"` → `\"` and breaks quote-matching
245
+ // patterns like no-credential-leak's `password = "..."`).
246
+ const content = Object.values(attack.params)
247
+ .filter((v): v is string => typeof v === 'string')
248
+ .join('\n')
249
+ const constitutionViolations = constitution.audit(content)
246
250
  if (constitutionViolations.length > 0) {
247
251
  return {
248
252
  attack,
package/src/index.tsx CHANGED
@@ -615,6 +615,16 @@ export async function runApp(options: RunOptions): Promise<void> {
615
615
  process.on('exit', () => {
616
616
  clearInterval(heartbeatInterval)
617
617
  unregisterSession(sessionName)
618
+ // Close the CRSI effectiveness loop on non-interactive exit paths
619
+ // (daemon worker / crash / kill) that bypass saveAndExit. saveAndExit
620
+ // already flushes; guard on !saved to avoid double-evaluating.
621
+ if (!saved) {
622
+ try {
623
+ engine.getAutoMemory().flushEffectiveness()
624
+ } catch {
625
+ // ignore — CRSI flush is non-critical
626
+ }
627
+ }
618
628
  if (!saved && context.getMessageCount() > 0) {
619
629
  SessionStore.autoSave(context.getMessages(), {
620
630
  provider: defaultProvider,
@@ -9,7 +9,7 @@
9
9
  export const PACKAGE_NAME = '@miphamai/cli' as const
10
10
 
11
11
  /** 当前发布版本 */
12
- export const PACKAGE_VERSION = '0.42.0' as const
12
+ export const PACKAGE_VERSION = '0.43.0' as const
13
13
 
14
14
  /** npm install 全局安装命令 */
15
15
  export const NPM_INSTALL_COMMAND = `npm install -g ${PACKAGE_NAME}` as const
@@ -10,6 +10,10 @@ import type { SkillsLoader } from '../skills/loader'
10
10
  import type { PluginManager } from '../plugin/plugin-manager'
11
11
  import type { Message } from '../shared/types.js'
12
12
  import { McpClient } from '../mcp/client'
13
+ import { buildCapabilityReport } from '../core/capability-inventory'
14
+ import { runCrsiModification, approvePending, rejectPending, hasPending } from '../core/crsi-modify'
15
+ import { produceCrsiProposal, LESSONS_FILE } from '../core/crsi-producer'
16
+ import { runEval, appendEvalScore } from '../core/eval-harness'
13
17
  import { NPM_UPDATE_COMMAND, PACKAGE_VERSION } from '../shared/index.ts'
14
18
  import { getPreference } from '../config/preferences'
15
19
  import { loadCrossSessionConfig } from '../config/loader'
@@ -731,6 +735,123 @@ const crsiStatsCmd: CommandHandler = async (ctx) => {
731
735
  return { content: lines.join('\n') }
732
736
  }
733
737
 
738
+ const crsiInventoryCmd: CommandHandler = async (ctx) => {
739
+ return { content: buildCapabilityReport(ctx.engine) }
740
+ }
741
+
742
+ const crsiModifyCmd: CommandHandler = async (ctx, args) => {
743
+ if (args[0] === '--approve') {
744
+ const r = approvePending()
745
+ return { content: r.success ? `✅ ${r.message}` : `⚠️ ${r.message}` }
746
+ }
747
+ if (args[0] === '--reject') {
748
+ const r = rejectPending()
749
+ return { content: r.success ? `✅ ${r.message}` : `⚠️ ${r.message}` }
750
+ }
751
+ if (args.length < 3) {
752
+ return {
753
+ content:
754
+ 'Usage: /crsi modify <description> <filePath> <newContent>\n' +
755
+ '- description 单 token(不含空格)\n' +
756
+ '- newContent 用 \\n 表示换行\n' +
757
+ '测试通过后:/crsi modify --approve 合并,/crsi modify --reject 丢弃',
758
+ }
759
+ }
760
+ if (hasPending()) {
761
+ return { content: '⚠️ 已有待批准的修改。先 /crsi modify --approve 或 --reject。' }
762
+ }
763
+
764
+ const description = args[0]!
765
+ const filePath = args[1]!
766
+ const newContent = args.slice(2).join(' ').replace(/\\n/g, '\n')
767
+
768
+ let originalContent = ''
769
+ try {
770
+ const { readFileSync } = await import('node:fs')
771
+ const { join } = await import('node:path')
772
+ originalContent = readFileSync(join(process.cwd(), filePath), 'utf-8')
773
+ } catch {
774
+ // 文件不存在 → 宽松模式(originalContent 为空)
775
+ }
776
+
777
+ const result = runCrsiModification({ description, filePath, newContent, originalContent })
778
+ if (!result.applied || result.phase === 'failed') {
779
+ return {
780
+ content: `❌ 修改未通过(phase: ${result.phase})。\n${result.error ?? ''}`,
781
+ }
782
+ }
783
+
784
+ return {
785
+ content:
786
+ `✅ 测试通过。审阅下方 diff:\n\n${result.diff}\n\n` +
787
+ '/crsi modify --approve 合并\n/crsi modify --reject 丢弃',
788
+ }
789
+ }
790
+
791
+ const crsiProposeCmd: CommandHandler = async (ctx) => {
792
+ if (hasPending()) {
793
+ return { content: '⚠️ 已有待批准的修改。先 /crsi modify --approve 或 --reject。' }
794
+ }
795
+
796
+ const insights = ctx.engine.getAutoMemory?.()?.accumulatedInsights ?? []
797
+ let metaRules: Parameters<typeof produceCrsiProposal>[1] = []
798
+ try {
799
+ metaRules = ctx.engine.getMetaRuleEngine?.()?.analyze().metaRules ?? []
800
+ } catch {
801
+ metaRules = []
802
+ }
803
+
804
+ // 教训文件按仓库根解析(沙箱的 filePath 是仓库根相对)。
805
+ let current = ''
806
+ try {
807
+ const { readFileSync } = await import('node:fs')
808
+ const { join } = await import('node:path')
809
+ const root = execSync('git rev-parse --show-toplevel', {
810
+ timeout: 5000,
811
+ encoding: 'utf-8',
812
+ }).trim()
813
+ current = readFileSync(join(root, LESSONS_FILE), 'utf-8')
814
+ } catch {
815
+ current = ''
816
+ }
817
+
818
+ const proposal = produceCrsiProposal(insights, metaRules, current, new Date().toISOString())
819
+
820
+ if (!proposal) {
821
+ return { content: '没有足够的失败信号(autoApplicable insight 或高置信元规则)来生成教训。' }
822
+ }
823
+
824
+ const result = runCrsiModification(proposal)
825
+ if (!result.applied || result.phase === 'failed') {
826
+ return { content: `❌ 生成失败(phase: ${result.phase})。\n${result.error ?? ''}` }
827
+ }
828
+
829
+ return {
830
+ content:
831
+ `✅ 已生成教训并跑过测试。审阅 diff:\n\n${result.diff}\n\n` +
832
+ '/crsi modify --approve 合并 | /crsi modify --reject 丢弃',
833
+ }
834
+ }
835
+
836
+ const crsiEvalCmd: CommandHandler = async () => {
837
+ const report = runEval()
838
+ appendEvalScore(report)
839
+
840
+ const lines: string[] = ['## 🧪 CRSI Eval Harness', '']
841
+ lines.push(`得分: **${report.score}/100** (${report.passed}/${report.total})`, '')
842
+ lines.push('| 任务 | 结果 |')
843
+ lines.push('|------|------|')
844
+ for (const r of report.results) {
845
+ lines.push(
846
+ `| ${r.description} | ${r.passed ? '✅' : '❌'}${r.detail ? ` — ${r.detail}` : ''} |`,
847
+ )
848
+ }
849
+ if (report.failures.length > 0) {
850
+ lines.push('', `❌ 失败任务: ${report.failures.join(', ')}`)
851
+ }
852
+ return { content: lines.join('\n') }
853
+ }
854
+
734
855
  const crsiHealthCmd: CommandHandler = async (ctx) => {
735
856
  const engine = ctx.engine.getRuleEngine()
736
857
  const tracker = ctx.engine.getEffectivenessTracker()
@@ -4313,6 +4434,10 @@ const commandsListCmd: CommandHandler = () => {
4313
4434
  '/crsi restore': 'Tools & Skills',
4314
4435
  '/crsi stats': 'Tools & Skills',
4315
4436
  '/crsi health': 'Tools & Skills',
4437
+ '/crsi inventory': 'Tools & Skills',
4438
+ '/crsi modify': 'Tools & Skills',
4439
+ '/crsi propose': 'Tools & Skills',
4440
+ '/crsi eval': 'Tools & Skills',
4316
4441
  '/crsi meta': 'Tools & Skills',
4317
4442
  '/crsi interpret': 'Tools & Skills',
4318
4443
  '/crsi critique': 'Tools & Skills',
@@ -4465,6 +4590,10 @@ registry.set('/crsi analyze', crsiAnalyzeCmd)
4465
4590
  registry.set('/crsi restore', crsiRestoreCmd)
4466
4591
  registry.set('/crsi stats', crsiStatsCmd)
4467
4592
  registry.set('/crsi health', crsiHealthCmd)
4593
+ registry.set('/crsi inventory', crsiInventoryCmd)
4594
+ registry.set('/crsi modify', crsiModifyCmd)
4595
+ registry.set('/crsi propose', crsiProposeCmd)
4596
+ registry.set('/crsi eval', crsiEvalCmd)
4468
4597
  registry.set('/crsi meta', crsiMetaCmd)
4469
4598
  registry.set('/crsi interpret', crsiInterpretCmd)
4470
4599
  registry.set('/crsi critique', crsiCritiqueCmd)
@@ -4640,6 +4769,10 @@ const COMMAND_DESCRIPTIONS: Record<string, string> = {
4640
4769
  '/crsi restore': 'Restore a disabled or degraded CRSI rule',
4641
4770
  '/crsi stats': 'Show CRSI overall effectiveness statistics',
4642
4771
  '/crsi health': 'CRSI + SIS unified health dashboard with scoring',
4772
+ '/crsi inventory': 'Live capability self-report — CRSI/SIS/constitution state',
4773
+ '/crsi modify': 'Run a code self-modification through the sandbox (worktree → tests → approve)',
4774
+ '/crsi propose': 'Produce a codified CRSI lesson from failure signals, gated by the sandbox',
4775
+ '/crsi eval': 'Run the ground-truth CRSI eval harness and record the score',
4643
4776
  '/crsi meta': 'RSI Level 3 meta-rule analysis — rules that improve the rules',
4644
4777
  '/crsi interpret': 'Tool-call behavior dashboard — error patterns, usage, health',
4645
4778
  '/crsi critique': 'Enable/disable RLAIF self-critique on tool calls',