@miphamai/cli 0.62.0 → 0.64.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,7 +19,7 @@ import { ConstitutionLoader, DEFAULT_CONSTITUTION } from './constitution-loader'
19
19
  import { ErrorSignatureDB } from './error-signature-db'
20
20
  import { PreFlightChecker } from './preflight-checker'
21
21
  import { RedTeam } from './red-team'
22
- import { isProtectedPath, validateBlastRadius } from './crsi-sandbox'
22
+ import { isProtectedPath, validateBlastRadius, PROTECTED_CRITICAL_FILES } from './crsi-sandbox'
23
23
  import { produceRuleProposal, MANAGED_RULES_FILE } from './crsi-producer'
24
24
  import type { CrsiSignal } from './crsi-producer'
25
25
  import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
@@ -46,13 +46,17 @@ export interface EvalReport {
46
46
 
47
47
  const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
48
48
 
49
- /** 追加一次评估分数到 rewards 日志。 */
50
- export function appendEvalScore(report: EvalReport): void {
49
+ /** 追加一次评估分数到 rewards 日志(按奖励函数名键控)。 */
50
+ export function appendEvalScore(
51
+ name: string,
52
+ report: { score: number; passed: number; total: number },
53
+ ): void {
51
54
  try {
52
55
  mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
53
56
  appendFileSync(
54
57
  SCORES_FILE,
55
58
  JSON.stringify({
59
+ name,
56
60
  timestamp: new Date().toISOString(),
57
61
  score: report.score,
58
62
  passed: report.passed,
@@ -65,14 +69,16 @@ export function appendEvalScore(report: EvalReport): void {
65
69
  }
66
70
  }
67
71
 
68
- /** 读取最近一次评估分数(无记录时返回 null)。 */
69
- export function getLastEvalScore(): number | null {
72
+ /** 读取某奖励函数最近一次分数(无记录时返回 null)。旧无 name 记录自然跳过。 */
73
+ export function getLastEvalScore(name: string): number | null {
70
74
  try {
71
75
  if (!existsSync(SCORES_FILE)) return null
72
76
  const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
73
- if (lines.length === 0) return null
74
- const last = JSON.parse(lines[lines.length - 1]!) as { score?: number }
75
- return typeof last.score === 'number' ? last.score : null
77
+ for (let i = lines.length - 1; i >= 0; i--) {
78
+ const rec = JSON.parse(lines[i]!) as { name?: string; score?: number }
79
+ if (rec.name === name && typeof rec.score === 'number') return rec.score
80
+ }
81
+ return null
76
82
  } catch {
77
83
  return null
78
84
  }
@@ -167,6 +173,15 @@ export function runEval(): EvalReport {
167
173
  results.push({ id, description: `受保护路径被拒: ${path}`, passed: isProtectedPath(path) })
168
174
  }
169
175
 
176
+ // ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
177
+ const unprotected = PROTECTED_CRITICAL_FILES.filter((f) => !isProtectedPath(f))
178
+ results.push({
179
+ id: 'protection-completeness',
180
+ description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
181
+ passed: unprotected.length === 0,
182
+ ...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
183
+ })
184
+
170
185
  // ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
171
186
  results.push({
172
187
  id: 'blast-radius-gate',
@@ -49,7 +49,8 @@ function executeCommand(cfg: HookConfig, ctx: HookContext): HookResult {
49
49
  }
50
50
 
51
51
  // Non-zero exit: check for block signal (exit code 2)
52
- const stderr = result.stderr?.toString() || ''
52
+ // 截断 stderr 防 MB 级 hook 输出溢出会话(对齐 HTTP hook 的 slice(0,2000))。
53
+ const stderr = (result.stderr?.toString() || '').slice(0, 2000)
53
54
 
54
55
  if (result.status === 2) {
55
56
  return {
@@ -0,0 +1,144 @@
1
+ // CRSI 改进轨:噪声自适应改进判定 + 台账 + pending verdict 闸。
2
+ // A1 不破:verdict / minEffect / 改进率全是确定性算术(均值/标准差/阈值/Wilson),无 LLM 裁判。
3
+ import { appendFileSync, readFileSync, existsSync, mkdirSync } from 'node:fs'
4
+ import { join } from 'node:path'
5
+ import { homedir } from 'node:os'
6
+ import type { SkillDeltaSample } from './task-performance'
7
+
8
+ export type ImprovementVerdict = 'improved' | 'regressed' | 'inconclusive'
9
+
10
+ export const MIN_EFFECT_FLOOR = 20
11
+ export const NOISE_K = 2
12
+ export const FALSE_POSITIVE_BASELINE = 0.05
13
+
14
+ export interface ImprovementReport {
15
+ skillName: string
16
+ changeSet: string[]
17
+ causal: boolean
18
+ baselineScores: number[]
19
+ postScores: number[]
20
+ deltaMean: number
21
+ noise: number
22
+ minEffect: number
23
+ verdict: ImprovementVerdict
24
+ }
25
+
26
+ export interface ImprovementRecord extends ImprovementReport {
27
+ id: string
28
+ timestamp: string
29
+ }
30
+
31
+ // ── 纯函数 ──
32
+
33
+ export function computeMinEffect(noise: number): number {
34
+ return Math.max(MIN_EFFECT_FLOOR, NOISE_K * noise)
35
+ }
36
+
37
+ export function classifyDelta(deltaMean: number, minEffect: number): ImprovementVerdict {
38
+ if (deltaMean <= -minEffect) return 'regressed'
39
+ if (deltaMean >= minEffect) return 'improved'
40
+ return 'inconclusive'
41
+ }
42
+
43
+ function mean(xs: number[]): number {
44
+ return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length
45
+ }
46
+
47
+ function stdDev(xs: number[]): number {
48
+ if (xs.length < 2) return 0
49
+ const m = mean(xs)
50
+ const variance = xs.reduce((a, b) => a + (b - m) ** 2, 0) / (xs.length - 1)
51
+ return Math.sqrt(variance)
52
+ }
53
+
54
+ export function buildImprovementReport(
55
+ sample: SkillDeltaSample,
56
+ changeSet: string[],
57
+ ): ImprovementReport {
58
+ const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
59
+ const noise = stdDev(sample.baselineScores)
60
+ const minEffect = computeMinEffect(noise)
61
+ const verdict = classifyDelta(deltaMean, minEffect)
62
+ return {
63
+ skillName: sample.skillName,
64
+ changeSet,
65
+ causal: changeSet.length === 1,
66
+ baselineScores: sample.baselineScores,
67
+ postScores: sample.postScores,
68
+ deltaMean,
69
+ noise,
70
+ minEffect,
71
+ verdict,
72
+ }
73
+ }
74
+
75
+ /** Wilson score 区间(z 默认 1.96 = 95%)。n=0 → {0,0}。 */
76
+ export function wilsonInterval(
77
+ improved: number,
78
+ total: number,
79
+ z = 1.96,
80
+ ): { lo: number; hi: number } {
81
+ if (total === 0) return { lo: 0, hi: 0 }
82
+ const p = improved / total
83
+ const n = total
84
+ const denom = 1 + (z * z) / n
85
+ const center = (p + (z * z) / (2 * n)) / denom
86
+ const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / denom
87
+ return { lo: center - half, hi: center + half }
88
+ }
89
+
90
+ export function improvementRate(records: ImprovementRecord[]): {
91
+ total: number
92
+ improved: number
93
+ rate: number
94
+ lo: number
95
+ hi: number
96
+ } {
97
+ const total = records.length
98
+ const improved = records.filter((r) => r.verdict === 'improved').length
99
+ const { lo, hi } = wilsonInterval(improved, total)
100
+ return { total, improved, rate: total === 0 ? 0 : improved / total, lo, hi }
101
+ }
102
+
103
+ /** 循环有效性:改进率 Wilson 95% CI 下界 > 假阳性基线(默认 5%)。 */
104
+ export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
105
+ const { lo } = improvementRate(records)
106
+ return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
107
+ }
108
+
109
+ // ── 台账 ──
110
+
111
+ export function improvementPath(): string {
112
+ return join(homedir(), '.mipham', 'crsi', 'improvements.jsonl')
113
+ }
114
+
115
+ export function appendImprovement(record: ImprovementRecord): void {
116
+ const file = improvementPath()
117
+ mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
118
+ appendFileSync(file, JSON.stringify(record) + '\n', 'utf-8')
119
+ }
120
+
121
+ export function readImprovements(): ImprovementRecord[] {
122
+ const file = improvementPath()
123
+ if (!existsSync(file)) return []
124
+ return readFileSync(file, 'utf-8')
125
+ .split('\n')
126
+ .filter((line) => line.trim() !== '')
127
+ .map((line) => JSON.parse(line) as ImprovementRecord)
128
+ }
129
+
130
+ // ── pending verdict 闸(倒退才拦) ──
131
+
132
+ let pendingVerdict: ImprovementVerdict | null = null
133
+
134
+ export function setPendingVerdict(v: ImprovementVerdict | null): void {
135
+ pendingVerdict = v
136
+ }
137
+
138
+ export function getPendingVerdict(): ImprovementVerdict | null {
139
+ return pendingVerdict
140
+ }
141
+
142
+ export function shouldBlockApproval(v: ImprovementVerdict): boolean {
143
+ return v === 'regressed'
144
+ }
@@ -0,0 +1,44 @@
1
+ // CRSI RewardFn 接口——reward function = policy→feedback 抽象。
2
+ // 统一「给一个 policy 打分」:机制哨兵(runEval)与任务表现(runTaskPerformance)
3
+ // 都 conform 成 RewardFn,自改进环的 verify 阶段可对任意奖励源比分数、判退化。
4
+ import type { Llm } from '../providers/llm'
5
+ import { runEval } from './eval-harness'
6
+ import { runTaskPerformance } from './task-performance'
7
+
8
+ /** 奖励函数统一输出的「分数」形状——所有 RewardFn 的 evaluate 都产出它。 */
9
+ export interface ScoreReport {
10
+ total: number
11
+ passed: number
12
+ score: number // 0-100
13
+ failures: string[]
14
+ }
15
+
16
+ /** 奖励函数(reward function = policy→feedback):一个具名、可替换的打分器。 */
17
+ export interface RewardFn {
18
+ name: string
19
+ description: string
20
+ evaluate(): Promise<ScoreReport> | ScoreReport
21
+ }
22
+
23
+ /** 机制哨兵:冻结契约评当前仓库机制代码(无 LLM,同步)。 */
24
+ export function mechanismSentinel(): RewardFn {
25
+ return {
26
+ name: 'mechanism-sentinel',
27
+ description: '冻结契约评当前仓库机制代码(无 LLM)',
28
+ evaluate: () => runEval(),
29
+ }
30
+ }
31
+
32
+ /** 任务表现:LLM 生成代码 + 冻结测试判定(有 LLM,异步)。 */
33
+ export function taskPerformanceRewardFn(llm: Llm): RewardFn {
34
+ return {
35
+ name: 'task-performance',
36
+ description: 'LLM 生成 + 冻结测试评 skill/通用任务',
37
+ evaluate: () => runTaskPerformance(llm),
38
+ }
39
+ }
40
+
41
+ /** 可枚举的奖励函数注册表。llm 缺省时只含无 LLM 的机制哨兵。 */
42
+ export function listRewardFns(llm?: Llm): RewardFn[] {
43
+ return [mechanismSentinel(), ...(llm ? [taskPerformanceRewardFn(llm)] : [])]
44
+ }
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Generate the session name (ID) used for cross-session tracking, inbox routing,
3
+ * and session persistence.
4
+ *
5
+ * Includes milliseconds so two sessions launched in the same second do not
6
+ * collide on the shared inbox directory / active-session registry. (2026-08-27
7
+ * review: the old `slice(0, 19)` dropped ms and let same-second launches share
8
+ * one inbox, leaking / misdelivering cross-session messages.)
9
+ */
10
+ export function generateSessionName(resume: string | undefined, now: Date = new Date()): string {
11
+ return resume || `session-${now.toISOString().replace(/[:.]/g, '-')}`
12
+ }
@@ -0,0 +1,42 @@
1
+ {
2
+ "version": 1,
3
+ "tasks": [
4
+ {
5
+ "id": "perf-impl-quicksort",
6
+ "category": "test-driven",
7
+ "prompt": "实现并导出 quicksort 函数:export function quicksort(arr: number[]): number[]。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
8
+ "testCode": "import { test, expect } from 'bun:test'\nimport { quicksort } from './solution'\n\ntest('sorts numbers', () => {\n expect(quicksort([3, 1, 4, 1, 5, 9, 2, 6])).toEqual([1, 1, 2, 3, 4, 5, 6, 9])\n expect(quicksort([])).toEqual([])\n expect(quicksort([7])).toEqual([7])\n})\n"
9
+ },
10
+ {
11
+ "id": "perf-impl-fibonacci",
12
+ "category": "test-driven",
13
+ "prompt": "实现并导出 fibonacci 函数:export function fibonacci(n: number): number。返回第 n 项(fibonacci(0)=0, fibonacci(1)=1)。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
14
+ "testCode": "import { test, expect } from 'bun:test'\nimport { fibonacci } from './solution'\n\ntest('computes fibonacci', () => {\n expect(fibonacci(0)).toBe(0)\n expect(fibonacci(1)).toBe(1)\n expect(fibonacci(10)).toBe(55)\n expect(fibonacci(20)).toBe(6765)\n})\n"
15
+ },
16
+ {
17
+ "id": "perf-impl-binary-search",
18
+ "category": "test-driven",
19
+ "prompt": "实现并导出 binarySearch 函数:export function binarySearch(arr: number[], target: number): number。返回 target 的下标,不存在返回 -1。假设 arr 已升序。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
20
+ "testCode": "import { test, expect } from 'bun:test'\nimport { binarySearch } from './solution'\n\ntest('finds target', () => {\n expect(binarySearch([1, 2, 3, 4, 5], 3)).toBe(2)\n expect(binarySearch([1, 2, 3, 4, 5], 1)).toBe(0)\n expect(binarySearch([1, 2, 3, 4, 5], 99)).toBe(-1)\n})\n"
21
+ },
22
+ {
23
+ "id": "perf-impl-reverse-string",
24
+ "category": "test-driven",
25
+ "prompt": "实现并导出 reverseString 函数:export function reverseString(s: string): string。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
26
+ "testCode": "import { test, expect } from 'bun:test'\nimport { reverseString } from './solution'\n\ntest('reverses', () => {\n expect(reverseString('hello')).toBe('olleh')\n expect(reverseString('')).toBe('')\n expect(reverseString('a')).toBe('a')\n})\n"
27
+ },
28
+ {
29
+ "id": "perf-impl-max",
30
+ "category": "test-driven",
31
+ "prompt": "实现并导出 maxOf 函数:export function maxOf(arr: number[]): number。返回数组最大值,空数组返回 -Infinity。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
32
+ "testCode": "import { test, expect } from 'bun:test'\nimport { maxOf } from './solution'\n\ntest('finds max', () => {\n expect(maxOf([1, 5, 3, 9, 2])).toBe(9)\n expect(maxOf([-1, -5, -3])).toBe(-1)\n expect(maxOf([])).toBe(-Infinity)\n})\n"
33
+ },
34
+ {
35
+ "id": "perf-safe-parse-positive",
36
+ "category": "test-driven",
37
+ "skill": "safe-coding",
38
+ "prompt": "实现并导出 parsePositiveNumber 函数:export function parsePositiveNumber(input: string): number。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
39
+ "testCode": "import { test, expect } from 'bun:test'\nimport { parsePositiveNumber } from './solution'\n\ntest('rejects invalid input', () => {\n expect(() => parsePositiveNumber(null as any)).toThrow(RangeError)\n expect(() => parsePositiveNumber('')).toThrow(RangeError)\n expect(() => parsePositiveNumber('abc')).toThrow(RangeError)\n})\ntest('parses valid', () => {\n expect(parsePositiveNumber('42')).toBe(42)\n})\n"
40
+ }
41
+ ]
42
+ }
@@ -0,0 +1,244 @@
1
+ // apps/cli/src/core/task-performance.ts
2
+ // CRSI 任务表现评估器:LLM 生成代码 + 冻结测试判定,输出随改动变化的分数。
3
+ import { execSync } from 'node:child_process'
4
+ import { mkdtempSync, writeFileSync, rmSync, readFileSync } from 'node:fs'
5
+ import { tmpdir } from 'node:os'
6
+ import { join } from 'node:path'
7
+ import tasksFile from './task-performance-tasks.json' with { type: 'json' }
8
+ import type { Llm } from '../providers/llm'
9
+ import { parseFrontmatter } from '../skills/loader'
10
+
11
+ export type PerformanceTaskCategory = 'test-driven' | 'bug-fix'
12
+
13
+ export interface PerformanceTask {
14
+ id: string
15
+ category: PerformanceTaskCategory
16
+ prompt: string
17
+ testCode: string
18
+ skill?: string // 绑定被测 skill 的名字(= skill frontmatter name)
19
+ }
20
+
21
+ export function loadPerformanceTasks(): PerformanceTask[] {
22
+ return tasksFile.tasks as PerformanceTask[]
23
+ }
24
+
25
+ export interface JudgeResult {
26
+ passed: boolean
27
+ detail?: string
28
+ }
29
+
30
+ const DEFAULT_TIMEOUT_MS = 5000
31
+
32
+ /** 把生成代码 + 冻结测试写入临时目录,子进程跑 `bun test`,exit 0 即 pass。 */
33
+ export function judgeGeneratedCode(
34
+ testCode: string,
35
+ solutionCode: string,
36
+ opts?: { timeoutMs?: number },
37
+ ): JudgeResult {
38
+ const dir = mkdtempSync(join(tmpdir(), 'mipham-task-perf-'))
39
+ try {
40
+ writeFileSync(join(dir, 'solution.ts'), solutionCode)
41
+ writeFileSync(join(dir, 'solution.test.ts'), testCode)
42
+ try {
43
+ execSync('bun test solution.test.ts', {
44
+ cwd: dir,
45
+ timeout: opts?.timeoutMs ?? DEFAULT_TIMEOUT_MS,
46
+ stdio: 'pipe',
47
+ })
48
+ return { passed: true }
49
+ } catch (e) {
50
+ const stderr = (e as { stderr?: Buffer }).stderr?.toString() ?? ''
51
+ return { passed: false, detail: stderr.slice(0, 300) }
52
+ }
53
+ } finally {
54
+ rmSync(dir, { recursive: true, force: true })
55
+ }
56
+ }
57
+
58
+ export interface TaskPerformanceResult {
59
+ id: string
60
+ description: string
61
+ passed: boolean
62
+ detail?: string
63
+ }
64
+
65
+ export interface TaskPerformanceReport {
66
+ total: number
67
+ passed: number
68
+ score: number
69
+ results: TaskPerformanceResult[]
70
+ failures: string[]
71
+ }
72
+
73
+ /** 剥掉 LLM 可能包裹的 markdown 代码块(```ts ... ```),拿到裸代码。 */
74
+ export function stripCodeFences(text: string): string {
75
+ const fenced = text.match(/```(?:ts|typescript|js|javascript)?\n([\s\S]*?)```/)
76
+ return (fenced?.[1] ?? text).trim()
77
+ }
78
+
79
+ async function collectGeneratedCode(
80
+ llm: Llm,
81
+ prompt: string,
82
+ systemPrompt?: string,
83
+ ): Promise<string> {
84
+ let text = ''
85
+ const req = {
86
+ model: '', // falsy → registry 回退到 active model
87
+ messages: [{ role: 'user' as const, content: prompt }],
88
+ temperature: 0, // 温度 0,近确定
89
+ ...(systemPrompt ? { systemPrompt } : {}),
90
+ }
91
+ for await (const chunk of llm.chat(req)) {
92
+ if (chunk.type === 'text' && chunk.content) text += chunk.content
93
+ }
94
+ return stripCodeFences(text)
95
+ }
96
+
97
+ export async function runTaskPerformance(
98
+ llm: Llm,
99
+ opts?: { timeoutMs?: number; skill?: { name: string; text: string } },
100
+ ): Promise<TaskPerformanceReport> {
101
+ const wanted = opts?.skill?.name
102
+ const tasks = loadPerformanceTasks().filter((t) => (t.skill ?? undefined) === wanted)
103
+ const results: TaskPerformanceResult[] = []
104
+ for (const task of tasks) {
105
+ const code = await collectGeneratedCode(llm, task.prompt, opts?.skill?.text)
106
+ if (!code) {
107
+ results.push({
108
+ id: task.id,
109
+ description: task.prompt,
110
+ passed: false,
111
+ detail: 'LLM 未生成代码',
112
+ })
113
+ continue
114
+ }
115
+ const verdict = judgeGeneratedCode(task.testCode, code, opts)
116
+ results.push({
117
+ id: task.id,
118
+ description: task.prompt,
119
+ passed: verdict.passed,
120
+ detail: verdict.detail,
121
+ })
122
+ }
123
+ const passed = results.filter((r) => r.passed).length
124
+ return {
125
+ total: results.length,
126
+ passed,
127
+ score: results.length > 0 ? Math.round((passed / results.length) * 100) : 100,
128
+ results,
129
+ failures: results.filter((r) => !r.passed).map((r) => r.id),
130
+ }
131
+ }
132
+
133
+ export interface SkillDelta {
134
+ skillName: string
135
+ baseline: TaskPerformanceReport
136
+ post: TaskPerformanceReport
137
+ delta: number
138
+ }
139
+
140
+ interface ResolvedSkillProposal {
141
+ skillName: string
142
+ baselineText: string
143
+ postText: string
144
+ }
145
+
146
+ /** 路径门:只认内置 skill 文件。 */
147
+ function isSkillFile(filePath: string): boolean {
148
+ return (
149
+ filePath.startsWith('apps/cli/skills/') &&
150
+ (filePath.endsWith('.SKILL.md') || filePath.endsWith('.mipham-skill.md'))
151
+ )
152
+ }
153
+
154
+ /** 解析改 skill 的 proposal → 名字 + 旧/新 body;非 skill / 无任务集 / 旧不可读 → null。 */
155
+ function resolveSkillProposal(proposal: {
156
+ filePath: string
157
+ originalContent?: string
158
+ newContent: string
159
+ }): ResolvedSkillProposal | null {
160
+ if (!isSkillFile(proposal.filePath)) return null
161
+
162
+ const newParsed = parseFrontmatter(proposal.newContent)
163
+ const skillName = typeof newParsed.data.name === 'string' ? newParsed.data.name : undefined
164
+ if (!skillName) return null
165
+
166
+ if (!loadPerformanceTasks().some((t) => t.skill === skillName)) return null
167
+
168
+ let baselineText: string
169
+ if (proposal.originalContent !== undefined) {
170
+ baselineText = parseFrontmatter(proposal.originalContent).content
171
+ } else {
172
+ try {
173
+ baselineText = parseFrontmatter(readFileSync(proposal.filePath, 'utf-8')).content
174
+ } catch {
175
+ return null // 旧 skill 不可读 → 无可量
176
+ }
177
+ }
178
+
179
+ return { skillName, baselineText, postText: newParsed.content }
180
+ }
181
+
182
+ /**
183
+ * 测一个改 skill 的 proposal 的任务表现 before/after delta。
184
+ * 返回 null:不是 skill 文件 / 无匹配任务集(无可量)。
185
+ * A1 不破:只调 runTaskPerformance(LLM 生成 + 冻结测试判定)。
186
+ */
187
+ export async function measureSkillDelta(
188
+ llm: Llm,
189
+ proposal: { filePath: string; originalContent?: string; newContent: string },
190
+ ): Promise<SkillDelta | null> {
191
+ const resolved = resolveSkillProposal(proposal)
192
+ if (!resolved) return null
193
+
194
+ const baseline = await runTaskPerformance(llm, {
195
+ skill: { name: resolved.skillName, text: resolved.baselineText },
196
+ })
197
+ const post = await runTaskPerformance(llm, {
198
+ skill: { name: resolved.skillName, text: resolved.postText },
199
+ })
200
+
201
+ return {
202
+ skillName: resolved.skillName,
203
+ baseline,
204
+ post,
205
+ delta: post.score - baseline.score,
206
+ }
207
+ }
208
+
209
+ export interface SkillDeltaSample {
210
+ skillName: string
211
+ baselineScores: number[]
212
+ postScores: number[]
213
+ }
214
+
215
+ /**
216
+ * 多次采样 before/after(每次都是「LLM 生成 → 冻结测试判定」),供改进轨估噪声。
217
+ * 返回 null 门与 measureSkillDelta 相同。k 默认 3。
218
+ */
219
+ export async function measureSkillDeltaRepeated(
220
+ llm: Llm,
221
+ proposal: { filePath: string; originalContent?: string; newContent: string },
222
+ opts?: { k?: number },
223
+ ): Promise<SkillDeltaSample | null> {
224
+ const resolved = resolveSkillProposal(proposal)
225
+ if (!resolved) return null
226
+
227
+ const k = opts?.k ?? 3
228
+ const baselineScores: number[] = []
229
+ const postScores: number[] = []
230
+ for (let i = 0; i < k; i++) {
231
+ const r = await runTaskPerformance(llm, {
232
+ skill: { name: resolved.skillName, text: resolved.baselineText },
233
+ })
234
+ baselineScores.push(r.score)
235
+ }
236
+ for (let i = 0; i < k; i++) {
237
+ const r = await runTaskPerformance(llm, {
238
+ skill: { name: resolved.skillName, text: resolved.postText },
239
+ })
240
+ postScores.push(r.score)
241
+ }
242
+
243
+ return { skillName: resolved.skillName, baselineScores, postScores }
244
+ }
@@ -845,7 +845,9 @@
845
845
  "esc_to_interrupt": "esc to interrupt",
846
846
  "shift_tab_cycle": "shift+tab",
847
847
  "left_for_agents": "ctrl+g agents to manage",
848
- "mipham_modes": "Mipham"
848
+ "mipham_modes": "Mipham",
849
+ "update_available": "Update available · v{version}",
850
+ "update_installed_restart": "Update installed · Restart to apply"
849
851
  },
850
852
  "wizard": {
851
853
  "header_subtitle": " — Configuration Wizard",
@@ -845,7 +845,9 @@
845
845
  "esc_to_interrupt": "Esc 中断",
846
846
  "shift_tab_cycle": "Shift+Tab",
847
847
  "left_for_agents": "ctrl+g 管理代理",
848
- "mipham_modes": "Mipham独有"
848
+ "mipham_modes": "Mipham独有",
849
+ "update_available": "有新版 · v{version}",
850
+ "update_installed_restart": "更新已安装 · 重启生效"
849
851
  },
850
852
  "wizard": {
851
853
  "header_subtitle": " — 配置向导",
package/src/index.tsx CHANGED
@@ -25,6 +25,7 @@ import { loadSessionMemories, getMemoryManager } from './core/memory/memory-load
25
25
  import { ContextManager } from './core/context'
26
26
  import { PrefixCacheTracker } from './core/context-token'
27
27
  import { QueryEngine } from './core/engine'
28
+ import { generateSessionName } from './core/session-name'
28
29
  import { ExperienceRuleEngine } from './core/rule-engine.js'
29
30
  import { SessionLog } from './core/session-log'
30
31
  import { SessionStore } from './core/session-store'
@@ -40,6 +41,7 @@ import { mountConstitution, createConstitution } from './core/constitution-seam'
40
41
  import { ConstitutionLoader } from './core/constitution-loader'
41
42
  import { McpClient } from './mcp/client'
42
43
  import { registerMcpServerTools, syncMcpToolsOnChange } from './mcp/registry'
44
+ import { formatMcpConnectFailures, type McpConnectFailure } from './mcp/connect-failures'
43
45
  import { AgentRegistry } from './agent/agent-registry'
44
46
  import { HookEngine } from './core/hooks'
45
47
  import { ArtifactServer } from './artifacts/server'
@@ -202,8 +204,8 @@ function SetupGate(props: SetupGateProps) {
202
204
  async function connectMcpServers(
203
205
  mcpServers: McpServerConfig[],
204
206
  tools: ReturnType<typeof createToolRegistry>,
205
- ): Promise<void> {
206
- if (mcpServers.length === 0) return
207
+ ): Promise<McpConnectFailure[]> {
208
+ if (mcpServers.length === 0) return []
207
209
  const mcp = McpClient.getInstance()
208
210
  // Wire runtime tool-list changes before connect so mid-connect updates land.
209
211
  syncMcpToolsOnChange(mcp, tools)
@@ -216,14 +218,17 @@ async function connectMcpServers(
216
218
  }
217
219
  }),
218
220
  )
221
+ const failures: McpConnectFailure[] = []
219
222
  for (let i = 0; i < results.length; i++) {
220
223
  const result = results[i]!
221
224
  if (result.status === 'rejected') {
222
- process.stderr.write(
223
- `[mcp] Failed to connect "${mcpServers[i]!.name}": ${String(result.reason)}\n`,
224
- )
225
+ const name = mcpServers[i]!.name
226
+ const reason = String(result.reason)
227
+ failures.push({ name, reason })
228
+ process.stderr.write(`[mcp] Failed to connect "${name}": ${reason}\n`)
225
229
  }
226
230
  }
231
+ return failures
227
232
  }
228
233
 
229
234
  export async function runApp(options: RunOptions): Promise<void> {
@@ -325,8 +330,7 @@ export async function runApp(options: RunOptions): Promise<void> {
325
330
  const pluginManager = new PluginManager()
326
331
 
327
332
  // Generate session name for tracking (used by /cd to persist cwd)
328
- const sessionName =
329
- options.resume || `session-${new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)}`
333
+ const sessionName = generateSessionName(options.resume)
330
334
 
331
335
  // Initialize context — restore saved session if available
332
336
  // Read the active model's context window for dynamic max-token sizing.
@@ -498,7 +502,7 @@ export async function runApp(options: RunOptions): Promise<void> {
498
502
  // Connect MCP servers lazily in the background: the UI renders immediately,
499
503
  // and each server's tools register into the shared registry as it connects.
500
504
  // A dead server (e.g. a 15s connect timeout) no longer blocks startup.
501
- void connectMcpServers(config.skills?.mcpServers ?? [], tools)
505
+ const mcpConnectPromise = connectMcpServers(config.skills?.mcpServers ?? [], tools)
502
506
 
503
507
  // Initialize hook engine — register skill-defined hooks
504
508
  const hookEngine = new HookEngine()
@@ -562,6 +566,14 @@ export async function runApp(options: RunOptions): Promise<void> {
562
566
 
563
567
  // ── Cross-session messaging: register session, heartbeat, shutdown ──
564
568
  engine.setSessionId(sessionName)
569
+
570
+ // Surface MCP connect failures to the model once they settle, so it knows
571
+ // those servers' tools are unavailable (rather than silently concluding the
572
+ // tools don't exist). Injected asynchronously — never blocks first paint.
573
+ void mcpConnectPromise.then((failures) => {
574
+ const notice = formatMcpConnectFailures(failures)
575
+ if (notice) engine.getContext().addMessage({ role: 'user', content: notice })
576
+ })
565
577
  // Human-readable, cross-session-addressable name. Defaults to the cwd basename
566
578
  // so `@mipham-code` can address this session; uniqueness is enforced against
567
579
  // other live sessions (suffix -2, -3, …). The id stays a stable `session-<ts>`.