@miphamai/cli 0.62.0 → 0.64.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/standard/safe-coding.SKILL.md +9 -0
- package/src/agent/effectiveness-tracker.ts +2 -6
- package/src/agent/message-bus.ts +10 -0
- package/src/agent/recoverable-failure.ts +16 -10
- package/src/agent/sub-agent.ts +15 -3
- package/src/config/defaults.ts +4 -0
- package/src/core/autocomplete.ts +64 -0
- package/src/core/crsi-modify.ts +13 -11
- package/src/core/crsi-producer.ts +128 -2
- package/src/core/crsi-sandbox.ts +83 -25
- package/src/core/engine.ts +30 -2
- package/src/core/eval-harness.ts +23 -8
- package/src/core/hooks-executor.ts +2 -1
- package/src/core/improvement-track.ts +144 -0
- package/src/core/reward-fn.ts +44 -0
- package/src/core/session-name.ts +12 -0
- package/src/core/task-performance-tasks.json +42 -0
- package/src/core/task-performance.ts +244 -0
- package/src/i18n-core/locales/en-US.json +3 -1
- package/src/i18n-core/locales/zh-CN.json +3 -1
- package/src/index.tsx +20 -8
- package/src/mcp/connect-failures.ts +15 -0
- package/src/plugin/claude-plugin.ts +1 -1
- package/src/plugin/plugin-validator.ts +1 -1
- package/src/shared/package-info.ts +1 -1
- package/src/shared/types.ts +7 -0
- package/src/shared/update.ts +41 -0
- package/src/skills/bundled-skills.ts +1 -0
- package/src/skills/loader.ts +1 -1
- package/src/ui/agent-footer.tsx +5 -1
- package/src/ui/app.tsx +71 -3
- package/src/ui/commands.ts +162 -8
- package/src/ui/input.tsx +71 -0
package/src/core/eval-harness.ts
CHANGED
|
@@ -19,7 +19,7 @@ import { ConstitutionLoader, DEFAULT_CONSTITUTION } from './constitution-loader'
|
|
|
19
19
|
import { ErrorSignatureDB } from './error-signature-db'
|
|
20
20
|
import { PreFlightChecker } from './preflight-checker'
|
|
21
21
|
import { RedTeam } from './red-team'
|
|
22
|
-
import { isProtectedPath, validateBlastRadius } from './crsi-sandbox'
|
|
22
|
+
import { isProtectedPath, validateBlastRadius, PROTECTED_CRITICAL_FILES } from './crsi-sandbox'
|
|
23
23
|
import { produceRuleProposal, MANAGED_RULES_FILE } from './crsi-producer'
|
|
24
24
|
import type { CrsiSignal } from './crsi-producer'
|
|
25
25
|
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
@@ -46,13 +46,17 @@ export interface EvalReport {
|
|
|
46
46
|
|
|
47
47
|
const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
|
|
48
48
|
|
|
49
|
-
/** 追加一次评估分数到 rewards
|
|
50
|
-
export function appendEvalScore(
|
|
49
|
+
/** 追加一次评估分数到 rewards 日志(按奖励函数名键控)。 */
|
|
50
|
+
export function appendEvalScore(
|
|
51
|
+
name: string,
|
|
52
|
+
report: { score: number; passed: number; total: number },
|
|
53
|
+
): void {
|
|
51
54
|
try {
|
|
52
55
|
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
53
56
|
appendFileSync(
|
|
54
57
|
SCORES_FILE,
|
|
55
58
|
JSON.stringify({
|
|
59
|
+
name,
|
|
56
60
|
timestamp: new Date().toISOString(),
|
|
57
61
|
score: report.score,
|
|
58
62
|
passed: report.passed,
|
|
@@ -65,14 +69,16 @@ export function appendEvalScore(report: EvalReport): void {
|
|
|
65
69
|
}
|
|
66
70
|
}
|
|
67
71
|
|
|
68
|
-
/**
|
|
69
|
-
export function getLastEvalScore(): number | null {
|
|
72
|
+
/** 读取某奖励函数最近一次分数(无记录时返回 null)。旧无 name 记录自然跳过。 */
|
|
73
|
+
export function getLastEvalScore(name: string): number | null {
|
|
70
74
|
try {
|
|
71
75
|
if (!existsSync(SCORES_FILE)) return null
|
|
72
76
|
const lines = readFileSync(SCORES_FILE, 'utf-8').trim().split('\n').filter(Boolean)
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
77
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
78
|
+
const rec = JSON.parse(lines[i]!) as { name?: string; score?: number }
|
|
79
|
+
if (rec.name === name && typeof rec.score === 'number') return rec.score
|
|
80
|
+
}
|
|
81
|
+
return null
|
|
76
82
|
} catch {
|
|
77
83
|
return null
|
|
78
84
|
}
|
|
@@ -167,6 +173,15 @@ export function runEval(): EvalReport {
|
|
|
167
173
|
results.push({ id, description: `受保护路径被拒: ${path}`, passed: isProtectedPath(path) })
|
|
168
174
|
}
|
|
169
175
|
|
|
176
|
+
// ── 语义边界完整性(ground truth:金丝雀关键机制文件全覆盖) ──
|
|
177
|
+
const unprotected = PROTECTED_CRITICAL_FILES.filter((f) => !isProtectedPath(f))
|
|
178
|
+
results.push({
|
|
179
|
+
id: 'protection-completeness',
|
|
180
|
+
description: '语义保护边界覆盖全部关键机制文件(评估器 + 核心机制)',
|
|
181
|
+
passed: unprotected.length === 0,
|
|
182
|
+
...(unprotected.length > 0 ? { detail: `未保护: ${unprotected.join(', ')}` } : {}),
|
|
183
|
+
})
|
|
184
|
+
|
|
170
185
|
// ── 完整覆盖闸(ground truth:未声明 blast radius 的 proposal 被 fail-closed 拒绝) ──
|
|
171
186
|
results.push({
|
|
172
187
|
id: 'blast-radius-gate',
|
|
@@ -49,7 +49,8 @@ function executeCommand(cfg: HookConfig, ctx: HookContext): HookResult {
|
|
|
49
49
|
}
|
|
50
50
|
|
|
51
51
|
// Non-zero exit: check for block signal (exit code 2)
|
|
52
|
-
|
|
52
|
+
// 截断 stderr 防 MB 级 hook 输出溢出会话(对齐 HTTP hook 的 slice(0,2000))。
|
|
53
|
+
const stderr = (result.stderr?.toString() || '').slice(0, 2000)
|
|
53
54
|
|
|
54
55
|
if (result.status === 2) {
|
|
55
56
|
return {
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
// CRSI 改进轨:噪声自适应改进判定 + 台账 + pending verdict 闸。
|
|
2
|
+
// A1 不破:verdict / minEffect / 改进率全是确定性算术(均值/标准差/阈值/Wilson),无 LLM 裁判。
|
|
3
|
+
import { appendFileSync, readFileSync, existsSync, mkdirSync } from 'node:fs'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import { homedir } from 'node:os'
|
|
6
|
+
import type { SkillDeltaSample } from './task-performance'
|
|
7
|
+
|
|
8
|
+
export type ImprovementVerdict = 'improved' | 'regressed' | 'inconclusive'
|
|
9
|
+
|
|
10
|
+
export const MIN_EFFECT_FLOOR = 20
|
|
11
|
+
export const NOISE_K = 2
|
|
12
|
+
export const FALSE_POSITIVE_BASELINE = 0.05
|
|
13
|
+
|
|
14
|
+
export interface ImprovementReport {
|
|
15
|
+
skillName: string
|
|
16
|
+
changeSet: string[]
|
|
17
|
+
causal: boolean
|
|
18
|
+
baselineScores: number[]
|
|
19
|
+
postScores: number[]
|
|
20
|
+
deltaMean: number
|
|
21
|
+
noise: number
|
|
22
|
+
minEffect: number
|
|
23
|
+
verdict: ImprovementVerdict
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface ImprovementRecord extends ImprovementReport {
|
|
27
|
+
id: string
|
|
28
|
+
timestamp: string
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// ── 纯函数 ──
|
|
32
|
+
|
|
33
|
+
export function computeMinEffect(noise: number): number {
|
|
34
|
+
return Math.max(MIN_EFFECT_FLOOR, NOISE_K * noise)
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function classifyDelta(deltaMean: number, minEffect: number): ImprovementVerdict {
|
|
38
|
+
if (deltaMean <= -minEffect) return 'regressed'
|
|
39
|
+
if (deltaMean >= minEffect) return 'improved'
|
|
40
|
+
return 'inconclusive'
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function mean(xs: number[]): number {
|
|
44
|
+
return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function stdDev(xs: number[]): number {
|
|
48
|
+
if (xs.length < 2) return 0
|
|
49
|
+
const m = mean(xs)
|
|
50
|
+
const variance = xs.reduce((a, b) => a + (b - m) ** 2, 0) / (xs.length - 1)
|
|
51
|
+
return Math.sqrt(variance)
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function buildImprovementReport(
|
|
55
|
+
sample: SkillDeltaSample,
|
|
56
|
+
changeSet: string[],
|
|
57
|
+
): ImprovementReport {
|
|
58
|
+
const deltaMean = mean(sample.postScores) - mean(sample.baselineScores)
|
|
59
|
+
const noise = stdDev(sample.baselineScores)
|
|
60
|
+
const minEffect = computeMinEffect(noise)
|
|
61
|
+
const verdict = classifyDelta(deltaMean, minEffect)
|
|
62
|
+
return {
|
|
63
|
+
skillName: sample.skillName,
|
|
64
|
+
changeSet,
|
|
65
|
+
causal: changeSet.length === 1,
|
|
66
|
+
baselineScores: sample.baselineScores,
|
|
67
|
+
postScores: sample.postScores,
|
|
68
|
+
deltaMean,
|
|
69
|
+
noise,
|
|
70
|
+
minEffect,
|
|
71
|
+
verdict,
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Wilson score 区间(z 默认 1.96 = 95%)。n=0 → {0,0}。 */
|
|
76
|
+
export function wilsonInterval(
|
|
77
|
+
improved: number,
|
|
78
|
+
total: number,
|
|
79
|
+
z = 1.96,
|
|
80
|
+
): { lo: number; hi: number } {
|
|
81
|
+
if (total === 0) return { lo: 0, hi: 0 }
|
|
82
|
+
const p = improved / total
|
|
83
|
+
const n = total
|
|
84
|
+
const denom = 1 + (z * z) / n
|
|
85
|
+
const center = (p + (z * z) / (2 * n)) / denom
|
|
86
|
+
const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / denom
|
|
87
|
+
return { lo: center - half, hi: center + half }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function improvementRate(records: ImprovementRecord[]): {
|
|
91
|
+
total: number
|
|
92
|
+
improved: number
|
|
93
|
+
rate: number
|
|
94
|
+
lo: number
|
|
95
|
+
hi: number
|
|
96
|
+
} {
|
|
97
|
+
const total = records.length
|
|
98
|
+
const improved = records.filter((r) => r.verdict === 'improved').length
|
|
99
|
+
const { lo, hi } = wilsonInterval(improved, total)
|
|
100
|
+
return { total, improved, rate: total === 0 ? 0 : improved / total, lo, hi }
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** 循环有效性:改进率 Wilson 95% CI 下界 > 假阳性基线(默认 5%)。 */
|
|
104
|
+
export function improvementSignalStrong(records: ImprovementRecord[]): boolean {
|
|
105
|
+
const { lo } = improvementRate(records)
|
|
106
|
+
return records.length > 0 && lo > FALSE_POSITIVE_BASELINE
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// ── 台账 ──
|
|
110
|
+
|
|
111
|
+
export function improvementPath(): string {
|
|
112
|
+
return join(homedir(), '.mipham', 'crsi', 'improvements.jsonl')
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export function appendImprovement(record: ImprovementRecord): void {
|
|
116
|
+
const file = improvementPath()
|
|
117
|
+
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
118
|
+
appendFileSync(file, JSON.stringify(record) + '\n', 'utf-8')
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export function readImprovements(): ImprovementRecord[] {
|
|
122
|
+
const file = improvementPath()
|
|
123
|
+
if (!existsSync(file)) return []
|
|
124
|
+
return readFileSync(file, 'utf-8')
|
|
125
|
+
.split('\n')
|
|
126
|
+
.filter((line) => line.trim() !== '')
|
|
127
|
+
.map((line) => JSON.parse(line) as ImprovementRecord)
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// ── pending verdict 闸(倒退才拦) ──
|
|
131
|
+
|
|
132
|
+
let pendingVerdict: ImprovementVerdict | null = null
|
|
133
|
+
|
|
134
|
+
export function setPendingVerdict(v: ImprovementVerdict | null): void {
|
|
135
|
+
pendingVerdict = v
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export function getPendingVerdict(): ImprovementVerdict | null {
|
|
139
|
+
return pendingVerdict
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
export function shouldBlockApproval(v: ImprovementVerdict): boolean {
|
|
143
|
+
return v === 'regressed'
|
|
144
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// CRSI RewardFn 接口——reward function = policy→feedback 抽象。
|
|
2
|
+
// 统一「给一个 policy 打分」:机制哨兵(runEval)与任务表现(runTaskPerformance)
|
|
3
|
+
// 都 conform 成 RewardFn,自改进环的 verify 阶段可对任意奖励源比分数、判退化。
|
|
4
|
+
import type { Llm } from '../providers/llm'
|
|
5
|
+
import { runEval } from './eval-harness'
|
|
6
|
+
import { runTaskPerformance } from './task-performance'
|
|
7
|
+
|
|
8
|
+
/** 奖励函数统一输出的「分数」形状——所有 RewardFn 的 evaluate 都产出它。 */
|
|
9
|
+
export interface ScoreReport {
|
|
10
|
+
total: number
|
|
11
|
+
passed: number
|
|
12
|
+
score: number // 0-100
|
|
13
|
+
failures: string[]
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/** 奖励函数(reward function = policy→feedback):一个具名、可替换的打分器。 */
|
|
17
|
+
export interface RewardFn {
|
|
18
|
+
name: string
|
|
19
|
+
description: string
|
|
20
|
+
evaluate(): Promise<ScoreReport> | ScoreReport
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** 机制哨兵:冻结契约评当前仓库机制代码(无 LLM,同步)。 */
|
|
24
|
+
export function mechanismSentinel(): RewardFn {
|
|
25
|
+
return {
|
|
26
|
+
name: 'mechanism-sentinel',
|
|
27
|
+
description: '冻结契约评当前仓库机制代码(无 LLM)',
|
|
28
|
+
evaluate: () => runEval(),
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** 任务表现:LLM 生成代码 + 冻结测试判定(有 LLM,异步)。 */
|
|
33
|
+
export function taskPerformanceRewardFn(llm: Llm): RewardFn {
|
|
34
|
+
return {
|
|
35
|
+
name: 'task-performance',
|
|
36
|
+
description: 'LLM 生成 + 冻结测试评 skill/通用任务',
|
|
37
|
+
evaluate: () => runTaskPerformance(llm),
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** 可枚举的奖励函数注册表。llm 缺省时只含无 LLM 的机制哨兵。 */
|
|
42
|
+
export function listRewardFns(llm?: Llm): RewardFn[] {
|
|
43
|
+
return [mechanismSentinel(), ...(llm ? [taskPerformanceRewardFn(llm)] : [])]
|
|
44
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Generate the session name (ID) used for cross-session tracking, inbox routing,
|
|
3
|
+
* and session persistence.
|
|
4
|
+
*
|
|
5
|
+
* Includes milliseconds so two sessions launched in the same second do not
|
|
6
|
+
* collide on the shared inbox directory / active-session registry. (2026-08-27
|
|
7
|
+
* review: the old `slice(0, 19)` dropped ms and let same-second launches share
|
|
8
|
+
* one inbox, leaking / misdelivering cross-session messages.)
|
|
9
|
+
*/
|
|
10
|
+
export function generateSessionName(resume: string | undefined, now: Date = new Date()): string {
|
|
11
|
+
return resume || `session-${now.toISOString().replace(/[:.]/g, '-')}`
|
|
12
|
+
}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"tasks": [
|
|
4
|
+
{
|
|
5
|
+
"id": "perf-impl-quicksort",
|
|
6
|
+
"category": "test-driven",
|
|
7
|
+
"prompt": "实现并导出 quicksort 函数:export function quicksort(arr: number[]): number[]。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
8
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { quicksort } from './solution'\n\ntest('sorts numbers', () => {\n expect(quicksort([3, 1, 4, 1, 5, 9, 2, 6])).toEqual([1, 1, 2, 3, 4, 5, 6, 9])\n expect(quicksort([])).toEqual([])\n expect(quicksort([7])).toEqual([7])\n})\n"
|
|
9
|
+
},
|
|
10
|
+
{
|
|
11
|
+
"id": "perf-impl-fibonacci",
|
|
12
|
+
"category": "test-driven",
|
|
13
|
+
"prompt": "实现并导出 fibonacci 函数:export function fibonacci(n: number): number。返回第 n 项(fibonacci(0)=0, fibonacci(1)=1)。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
14
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { fibonacci } from './solution'\n\ntest('computes fibonacci', () => {\n expect(fibonacci(0)).toBe(0)\n expect(fibonacci(1)).toBe(1)\n expect(fibonacci(10)).toBe(55)\n expect(fibonacci(20)).toBe(6765)\n})\n"
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": "perf-impl-binary-search",
|
|
18
|
+
"category": "test-driven",
|
|
19
|
+
"prompt": "实现并导出 binarySearch 函数:export function binarySearch(arr: number[], target: number): number。返回 target 的下标,不存在返回 -1。假设 arr 已升序。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
20
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { binarySearch } from './solution'\n\ntest('finds target', () => {\n expect(binarySearch([1, 2, 3, 4, 5], 3)).toBe(2)\n expect(binarySearch([1, 2, 3, 4, 5], 1)).toBe(0)\n expect(binarySearch([1, 2, 3, 4, 5], 99)).toBe(-1)\n})\n"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"id": "perf-impl-reverse-string",
|
|
24
|
+
"category": "test-driven",
|
|
25
|
+
"prompt": "实现并导出 reverseString 函数:export function reverseString(s: string): string。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
26
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { reverseString } from './solution'\n\ntest('reverses', () => {\n expect(reverseString('hello')).toBe('olleh')\n expect(reverseString('')).toBe('')\n expect(reverseString('a')).toBe('a')\n})\n"
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"id": "perf-impl-max",
|
|
30
|
+
"category": "test-driven",
|
|
31
|
+
"prompt": "实现并导出 maxOf 函数:export function maxOf(arr: number[]): number。返回数组最大值,空数组返回 -Infinity。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
32
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { maxOf } from './solution'\n\ntest('finds max', () => {\n expect(maxOf([1, 5, 3, 9, 2])).toBe(9)\n expect(maxOf([-1, -5, -3])).toBe(-1)\n expect(maxOf([])).toBe(-Infinity)\n})\n"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "perf-safe-parse-positive",
|
|
36
|
+
"category": "test-driven",
|
|
37
|
+
"skill": "safe-coding",
|
|
38
|
+
"prompt": "实现并导出 parsePositiveNumber 函数:export function parsePositiveNumber(input: string): number。只输出 TypeScript 代码,不要解释、不要 markdown 代码块。",
|
|
39
|
+
"testCode": "import { test, expect } from 'bun:test'\nimport { parsePositiveNumber } from './solution'\n\ntest('rejects invalid input', () => {\n expect(() => parsePositiveNumber(null as any)).toThrow(RangeError)\n expect(() => parsePositiveNumber('')).toThrow(RangeError)\n expect(() => parsePositiveNumber('abc')).toThrow(RangeError)\n})\ntest('parses valid', () => {\n expect(parsePositiveNumber('42')).toBe(42)\n})\n"
|
|
40
|
+
}
|
|
41
|
+
]
|
|
42
|
+
}
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
// apps/cli/src/core/task-performance.ts
|
|
2
|
+
// CRSI 任务表现评估器:LLM 生成代码 + 冻结测试判定,输出随改动变化的分数。
|
|
3
|
+
import { execSync } from 'node:child_process'
|
|
4
|
+
import { mkdtempSync, writeFileSync, rmSync, readFileSync } from 'node:fs'
|
|
5
|
+
import { tmpdir } from 'node:os'
|
|
6
|
+
import { join } from 'node:path'
|
|
7
|
+
import tasksFile from './task-performance-tasks.json' with { type: 'json' }
|
|
8
|
+
import type { Llm } from '../providers/llm'
|
|
9
|
+
import { parseFrontmatter } from '../skills/loader'
|
|
10
|
+
|
|
11
|
+
export type PerformanceTaskCategory = 'test-driven' | 'bug-fix'
|
|
12
|
+
|
|
13
|
+
export interface PerformanceTask {
|
|
14
|
+
id: string
|
|
15
|
+
category: PerformanceTaskCategory
|
|
16
|
+
prompt: string
|
|
17
|
+
testCode: string
|
|
18
|
+
skill?: string // 绑定被测 skill 的名字(= skill frontmatter name)
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function loadPerformanceTasks(): PerformanceTask[] {
|
|
22
|
+
return tasksFile.tasks as PerformanceTask[]
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export interface JudgeResult {
|
|
26
|
+
passed: boolean
|
|
27
|
+
detail?: string
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const DEFAULT_TIMEOUT_MS = 5000
|
|
31
|
+
|
|
32
|
+
/** 把生成代码 + 冻结测试写入临时目录,子进程跑 `bun test`,exit 0 即 pass。 */
|
|
33
|
+
export function judgeGeneratedCode(
|
|
34
|
+
testCode: string,
|
|
35
|
+
solutionCode: string,
|
|
36
|
+
opts?: { timeoutMs?: number },
|
|
37
|
+
): JudgeResult {
|
|
38
|
+
const dir = mkdtempSync(join(tmpdir(), 'mipham-task-perf-'))
|
|
39
|
+
try {
|
|
40
|
+
writeFileSync(join(dir, 'solution.ts'), solutionCode)
|
|
41
|
+
writeFileSync(join(dir, 'solution.test.ts'), testCode)
|
|
42
|
+
try {
|
|
43
|
+
execSync('bun test solution.test.ts', {
|
|
44
|
+
cwd: dir,
|
|
45
|
+
timeout: opts?.timeoutMs ?? DEFAULT_TIMEOUT_MS,
|
|
46
|
+
stdio: 'pipe',
|
|
47
|
+
})
|
|
48
|
+
return { passed: true }
|
|
49
|
+
} catch (e) {
|
|
50
|
+
const stderr = (e as { stderr?: Buffer }).stderr?.toString() ?? ''
|
|
51
|
+
return { passed: false, detail: stderr.slice(0, 300) }
|
|
52
|
+
}
|
|
53
|
+
} finally {
|
|
54
|
+
rmSync(dir, { recursive: true, force: true })
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export interface TaskPerformanceResult {
|
|
59
|
+
id: string
|
|
60
|
+
description: string
|
|
61
|
+
passed: boolean
|
|
62
|
+
detail?: string
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export interface TaskPerformanceReport {
|
|
66
|
+
total: number
|
|
67
|
+
passed: number
|
|
68
|
+
score: number
|
|
69
|
+
results: TaskPerformanceResult[]
|
|
70
|
+
failures: string[]
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** 剥掉 LLM 可能包裹的 markdown 代码块(```ts ... ```),拿到裸代码。 */
|
|
74
|
+
export function stripCodeFences(text: string): string {
|
|
75
|
+
const fenced = text.match(/```(?:ts|typescript|js|javascript)?\n([\s\S]*?)```/)
|
|
76
|
+
return (fenced?.[1] ?? text).trim()
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
async function collectGeneratedCode(
|
|
80
|
+
llm: Llm,
|
|
81
|
+
prompt: string,
|
|
82
|
+
systemPrompt?: string,
|
|
83
|
+
): Promise<string> {
|
|
84
|
+
let text = ''
|
|
85
|
+
const req = {
|
|
86
|
+
model: '', // falsy → registry 回退到 active model
|
|
87
|
+
messages: [{ role: 'user' as const, content: prompt }],
|
|
88
|
+
temperature: 0, // 温度 0,近确定
|
|
89
|
+
...(systemPrompt ? { systemPrompt } : {}),
|
|
90
|
+
}
|
|
91
|
+
for await (const chunk of llm.chat(req)) {
|
|
92
|
+
if (chunk.type === 'text' && chunk.content) text += chunk.content
|
|
93
|
+
}
|
|
94
|
+
return stripCodeFences(text)
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export async function runTaskPerformance(
|
|
98
|
+
llm: Llm,
|
|
99
|
+
opts?: { timeoutMs?: number; skill?: { name: string; text: string } },
|
|
100
|
+
): Promise<TaskPerformanceReport> {
|
|
101
|
+
const wanted = opts?.skill?.name
|
|
102
|
+
const tasks = loadPerformanceTasks().filter((t) => (t.skill ?? undefined) === wanted)
|
|
103
|
+
const results: TaskPerformanceResult[] = []
|
|
104
|
+
for (const task of tasks) {
|
|
105
|
+
const code = await collectGeneratedCode(llm, task.prompt, opts?.skill?.text)
|
|
106
|
+
if (!code) {
|
|
107
|
+
results.push({
|
|
108
|
+
id: task.id,
|
|
109
|
+
description: task.prompt,
|
|
110
|
+
passed: false,
|
|
111
|
+
detail: 'LLM 未生成代码',
|
|
112
|
+
})
|
|
113
|
+
continue
|
|
114
|
+
}
|
|
115
|
+
const verdict = judgeGeneratedCode(task.testCode, code, opts)
|
|
116
|
+
results.push({
|
|
117
|
+
id: task.id,
|
|
118
|
+
description: task.prompt,
|
|
119
|
+
passed: verdict.passed,
|
|
120
|
+
detail: verdict.detail,
|
|
121
|
+
})
|
|
122
|
+
}
|
|
123
|
+
const passed = results.filter((r) => r.passed).length
|
|
124
|
+
return {
|
|
125
|
+
total: results.length,
|
|
126
|
+
passed,
|
|
127
|
+
score: results.length > 0 ? Math.round((passed / results.length) * 100) : 100,
|
|
128
|
+
results,
|
|
129
|
+
failures: results.filter((r) => !r.passed).map((r) => r.id),
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
export interface SkillDelta {
|
|
134
|
+
skillName: string
|
|
135
|
+
baseline: TaskPerformanceReport
|
|
136
|
+
post: TaskPerformanceReport
|
|
137
|
+
delta: number
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
interface ResolvedSkillProposal {
|
|
141
|
+
skillName: string
|
|
142
|
+
baselineText: string
|
|
143
|
+
postText: string
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** 路径门:只认内置 skill 文件。 */
|
|
147
|
+
function isSkillFile(filePath: string): boolean {
|
|
148
|
+
return (
|
|
149
|
+
filePath.startsWith('apps/cli/skills/') &&
|
|
150
|
+
(filePath.endsWith('.SKILL.md') || filePath.endsWith('.mipham-skill.md'))
|
|
151
|
+
)
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** 解析改 skill 的 proposal → 名字 + 旧/新 body;非 skill / 无任务集 / 旧不可读 → null。 */
|
|
155
|
+
function resolveSkillProposal(proposal: {
|
|
156
|
+
filePath: string
|
|
157
|
+
originalContent?: string
|
|
158
|
+
newContent: string
|
|
159
|
+
}): ResolvedSkillProposal | null {
|
|
160
|
+
if (!isSkillFile(proposal.filePath)) return null
|
|
161
|
+
|
|
162
|
+
const newParsed = parseFrontmatter(proposal.newContent)
|
|
163
|
+
const skillName = typeof newParsed.data.name === 'string' ? newParsed.data.name : undefined
|
|
164
|
+
if (!skillName) return null
|
|
165
|
+
|
|
166
|
+
if (!loadPerformanceTasks().some((t) => t.skill === skillName)) return null
|
|
167
|
+
|
|
168
|
+
let baselineText: string
|
|
169
|
+
if (proposal.originalContent !== undefined) {
|
|
170
|
+
baselineText = parseFrontmatter(proposal.originalContent).content
|
|
171
|
+
} else {
|
|
172
|
+
try {
|
|
173
|
+
baselineText = parseFrontmatter(readFileSync(proposal.filePath, 'utf-8')).content
|
|
174
|
+
} catch {
|
|
175
|
+
return null // 旧 skill 不可读 → 无可量
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
return { skillName, baselineText, postText: newParsed.content }
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* 测一个改 skill 的 proposal 的任务表现 before/after delta。
|
|
184
|
+
* 返回 null:不是 skill 文件 / 无匹配任务集(无可量)。
|
|
185
|
+
* A1 不破:只调 runTaskPerformance(LLM 生成 + 冻结测试判定)。
|
|
186
|
+
*/
|
|
187
|
+
export async function measureSkillDelta(
|
|
188
|
+
llm: Llm,
|
|
189
|
+
proposal: { filePath: string; originalContent?: string; newContent: string },
|
|
190
|
+
): Promise<SkillDelta | null> {
|
|
191
|
+
const resolved = resolveSkillProposal(proposal)
|
|
192
|
+
if (!resolved) return null
|
|
193
|
+
|
|
194
|
+
const baseline = await runTaskPerformance(llm, {
|
|
195
|
+
skill: { name: resolved.skillName, text: resolved.baselineText },
|
|
196
|
+
})
|
|
197
|
+
const post = await runTaskPerformance(llm, {
|
|
198
|
+
skill: { name: resolved.skillName, text: resolved.postText },
|
|
199
|
+
})
|
|
200
|
+
|
|
201
|
+
return {
|
|
202
|
+
skillName: resolved.skillName,
|
|
203
|
+
baseline,
|
|
204
|
+
post,
|
|
205
|
+
delta: post.score - baseline.score,
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
export interface SkillDeltaSample {
|
|
210
|
+
skillName: string
|
|
211
|
+
baselineScores: number[]
|
|
212
|
+
postScores: number[]
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* 多次采样 before/after(每次都是「LLM 生成 → 冻结测试判定」),供改进轨估噪声。
|
|
217
|
+
* 返回 null 门与 measureSkillDelta 相同。k 默认 3。
|
|
218
|
+
*/
|
|
219
|
+
export async function measureSkillDeltaRepeated(
|
|
220
|
+
llm: Llm,
|
|
221
|
+
proposal: { filePath: string; originalContent?: string; newContent: string },
|
|
222
|
+
opts?: { k?: number },
|
|
223
|
+
): Promise<SkillDeltaSample | null> {
|
|
224
|
+
const resolved = resolveSkillProposal(proposal)
|
|
225
|
+
if (!resolved) return null
|
|
226
|
+
|
|
227
|
+
const k = opts?.k ?? 3
|
|
228
|
+
const baselineScores: number[] = []
|
|
229
|
+
const postScores: number[] = []
|
|
230
|
+
for (let i = 0; i < k; i++) {
|
|
231
|
+
const r = await runTaskPerformance(llm, {
|
|
232
|
+
skill: { name: resolved.skillName, text: resolved.baselineText },
|
|
233
|
+
})
|
|
234
|
+
baselineScores.push(r.score)
|
|
235
|
+
}
|
|
236
|
+
for (let i = 0; i < k; i++) {
|
|
237
|
+
const r = await runTaskPerformance(llm, {
|
|
238
|
+
skill: { name: resolved.skillName, text: resolved.postText },
|
|
239
|
+
})
|
|
240
|
+
postScores.push(r.score)
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
return { skillName: resolved.skillName, baselineScores, postScores }
|
|
244
|
+
}
|
|
@@ -845,7 +845,9 @@
|
|
|
845
845
|
"esc_to_interrupt": "esc to interrupt",
|
|
846
846
|
"shift_tab_cycle": "shift+tab",
|
|
847
847
|
"left_for_agents": "ctrl+g agents to manage",
|
|
848
|
-
"mipham_modes": "Mipham"
|
|
848
|
+
"mipham_modes": "Mipham",
|
|
849
|
+
"update_available": "Update available · v{version}",
|
|
850
|
+
"update_installed_restart": "Update installed · Restart to apply"
|
|
849
851
|
},
|
|
850
852
|
"wizard": {
|
|
851
853
|
"header_subtitle": " — Configuration Wizard",
|
|
@@ -845,7 +845,9 @@
|
|
|
845
845
|
"esc_to_interrupt": "Esc 中断",
|
|
846
846
|
"shift_tab_cycle": "Shift+Tab",
|
|
847
847
|
"left_for_agents": "ctrl+g 管理代理",
|
|
848
|
-
"mipham_modes": "Mipham独有"
|
|
848
|
+
"mipham_modes": "Mipham独有",
|
|
849
|
+
"update_available": "有新版 · v{version}",
|
|
850
|
+
"update_installed_restart": "更新已安装 · 重启生效"
|
|
849
851
|
},
|
|
850
852
|
"wizard": {
|
|
851
853
|
"header_subtitle": " — 配置向导",
|
package/src/index.tsx
CHANGED
|
@@ -25,6 +25,7 @@ import { loadSessionMemories, getMemoryManager } from './core/memory/memory-load
|
|
|
25
25
|
import { ContextManager } from './core/context'
|
|
26
26
|
import { PrefixCacheTracker } from './core/context-token'
|
|
27
27
|
import { QueryEngine } from './core/engine'
|
|
28
|
+
import { generateSessionName } from './core/session-name'
|
|
28
29
|
import { ExperienceRuleEngine } from './core/rule-engine.js'
|
|
29
30
|
import { SessionLog } from './core/session-log'
|
|
30
31
|
import { SessionStore } from './core/session-store'
|
|
@@ -40,6 +41,7 @@ import { mountConstitution, createConstitution } from './core/constitution-seam'
|
|
|
40
41
|
import { ConstitutionLoader } from './core/constitution-loader'
|
|
41
42
|
import { McpClient } from './mcp/client'
|
|
42
43
|
import { registerMcpServerTools, syncMcpToolsOnChange } from './mcp/registry'
|
|
44
|
+
import { formatMcpConnectFailures, type McpConnectFailure } from './mcp/connect-failures'
|
|
43
45
|
import { AgentRegistry } from './agent/agent-registry'
|
|
44
46
|
import { HookEngine } from './core/hooks'
|
|
45
47
|
import { ArtifactServer } from './artifacts/server'
|
|
@@ -202,8 +204,8 @@ function SetupGate(props: SetupGateProps) {
|
|
|
202
204
|
async function connectMcpServers(
|
|
203
205
|
mcpServers: McpServerConfig[],
|
|
204
206
|
tools: ReturnType<typeof createToolRegistry>,
|
|
205
|
-
): Promise<
|
|
206
|
-
if (mcpServers.length === 0) return
|
|
207
|
+
): Promise<McpConnectFailure[]> {
|
|
208
|
+
if (mcpServers.length === 0) return []
|
|
207
209
|
const mcp = McpClient.getInstance()
|
|
208
210
|
// Wire runtime tool-list changes before connect so mid-connect updates land.
|
|
209
211
|
syncMcpToolsOnChange(mcp, tools)
|
|
@@ -216,14 +218,17 @@ async function connectMcpServers(
|
|
|
216
218
|
}
|
|
217
219
|
}),
|
|
218
220
|
)
|
|
221
|
+
const failures: McpConnectFailure[] = []
|
|
219
222
|
for (let i = 0; i < results.length; i++) {
|
|
220
223
|
const result = results[i]!
|
|
221
224
|
if (result.status === 'rejected') {
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
)
|
|
225
|
+
const name = mcpServers[i]!.name
|
|
226
|
+
const reason = String(result.reason)
|
|
227
|
+
failures.push({ name, reason })
|
|
228
|
+
process.stderr.write(`[mcp] Failed to connect "${name}": ${reason}\n`)
|
|
225
229
|
}
|
|
226
230
|
}
|
|
231
|
+
return failures
|
|
227
232
|
}
|
|
228
233
|
|
|
229
234
|
export async function runApp(options: RunOptions): Promise<void> {
|
|
@@ -325,8 +330,7 @@ export async function runApp(options: RunOptions): Promise<void> {
|
|
|
325
330
|
const pluginManager = new PluginManager()
|
|
326
331
|
|
|
327
332
|
// Generate session name for tracking (used by /cd to persist cwd)
|
|
328
|
-
const sessionName =
|
|
329
|
-
options.resume || `session-${new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)}`
|
|
333
|
+
const sessionName = generateSessionName(options.resume)
|
|
330
334
|
|
|
331
335
|
// Initialize context — restore saved session if available
|
|
332
336
|
// Read the active model's context window for dynamic max-token sizing.
|
|
@@ -498,7 +502,7 @@ export async function runApp(options: RunOptions): Promise<void> {
|
|
|
498
502
|
// Connect MCP servers lazily in the background: the UI renders immediately,
|
|
499
503
|
// and each server's tools register into the shared registry as it connects.
|
|
500
504
|
// A dead server (e.g. a 15s connect timeout) no longer blocks startup.
|
|
501
|
-
|
|
505
|
+
const mcpConnectPromise = connectMcpServers(config.skills?.mcpServers ?? [], tools)
|
|
502
506
|
|
|
503
507
|
// Initialize hook engine — register skill-defined hooks
|
|
504
508
|
const hookEngine = new HookEngine()
|
|
@@ -562,6 +566,14 @@ export async function runApp(options: RunOptions): Promise<void> {
|
|
|
562
566
|
|
|
563
567
|
// ── Cross-session messaging: register session, heartbeat, shutdown ──
|
|
564
568
|
engine.setSessionId(sessionName)
|
|
569
|
+
|
|
570
|
+
// Surface MCP connect failures to the model once they settle, so it knows
|
|
571
|
+
// those servers' tools are unavailable (rather than silently concluding the
|
|
572
|
+
// tools don't exist). Injected asynchronously — never blocks first paint.
|
|
573
|
+
void mcpConnectPromise.then((failures) => {
|
|
574
|
+
const notice = formatMcpConnectFailures(failures)
|
|
575
|
+
if (notice) engine.getContext().addMessage({ role: 'user', content: notice })
|
|
576
|
+
})
|
|
565
577
|
// Human-readable, cross-session-addressable name. Defaults to the cwd basename
|
|
566
578
|
// so `@mipham-code` can address this session; uniqueness is enforced against
|
|
567
579
|
// other live sessions (suffix -2, -3, …). The id stays a stable `session-<ts>`.
|