@miphamai/cli 0.48.0 → 0.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/mipham/doc-sync.mipham-skill.md +198 -0
- package/skills/mipham/self-audit.mipham-skill.md +1 -1
- package/skills/standard/github-ops.SKILL.md +1 -1
- package/skills/standard/web-access/references/cdp-api.md +132 -0
- package/skills/standard/web-access/references/site-patterns/.gitkeep +0 -0
- package/skills/standard/web-access/scripts/cdp-proxy.mjs +756 -0
- package/skills/standard/web-access/scripts/check-deps.mjs +187 -0
- package/skills/standard/web-access/scripts/find-url.mjs +271 -0
- package/skills/standard/web-access/scripts/match-site.mjs +48 -0
- package/skills/standard/web-access.SKILL.md +77 -158
- package/src/agent/cross-session/discovery.ts +26 -1
- package/src/config/defaults.ts +3 -0
- package/src/core/behavior-tasks.json +101 -0
- package/src/core/behavior-tasks.ts +44 -0
- package/src/core/crsi-producer.ts +214 -0
- package/src/core/crsi-sandbox.ts +5 -0
- package/src/core/eval-harness.ts +7 -0
- package/src/core/instructions.ts +11 -0
- package/src/core/proposal-guard.ts +100 -0
- package/src/core/task-runner-tasks.json +14 -0
- package/src/core/task-runner.ts +163 -0
- package/src/daemon/heartbeat.ts +83 -0
- package/src/daemon/server.ts +34 -2
- package/src/shared/package-info.ts +4 -1
- package/src/skills/bundled-skill-assets.ts +19 -0
- package/src/skills/bundled-skills.ts +4 -3
- package/src/skills/skill-assets.ts +37 -0
- package/src/tools/agent/skill.ts +13 -2
- package/src/tools/file/grep.ts +11 -1
- package/src/ui/app.tsx +21 -1
- package/src/ui/commands.ts +81 -6
|
@@ -12,6 +12,10 @@
|
|
|
12
12
|
|
|
13
13
|
import type { CrsiInsight } from './auto-memory'
|
|
14
14
|
import type { MetaRule } from './meta-rule-engine'
|
|
15
|
+
import type { Llm } from '../providers/llm'
|
|
16
|
+
import { readdirSync, appendFileSync, readFileSync, existsSync, mkdirSync, rmSync } from 'node:fs'
|
|
17
|
+
import { join } from 'node:path'
|
|
18
|
+
import { homedir } from 'node:os'
|
|
15
19
|
|
|
16
20
|
/** 教训文件(相对仓库根)。预建,沙箱只能改已存在文件。 */
|
|
17
21
|
export const LESSONS_FILE = 'apps/cli/crsi-lessons.md'
|
|
@@ -86,6 +90,9 @@ export function produceCrsiProposal(
|
|
|
86
90
|
const signal = selectCrsiSignal(insights, metaRules)
|
|
87
91
|
if (!signal) return null
|
|
88
92
|
|
|
93
|
+
// 幂等:同一信号的教训标题已在文件中,不再重复产出。
|
|
94
|
+
if (currentLessons.includes(`## ${signal.category}: ${signal.title}`)) return null
|
|
95
|
+
|
|
89
96
|
const lesson = buildLessonContent(signal, timestamp)
|
|
90
97
|
const newContent = currentLessons ? `${currentLessons.trimEnd()}\n\n${lesson}\n` : `${lesson}\n`
|
|
91
98
|
|
|
@@ -186,3 +193,210 @@ export function produceRuleProposal(
|
|
|
186
193
|
originalContent: currentManagedRules,
|
|
187
194
|
}
|
|
188
195
|
}
|
|
196
|
+
|
|
197
|
+
// ── Producer 散文提议(块 1):从失败信号生成「改 skill 散文」提议 ──
|
|
198
|
+
// A1 边界首次实演:LLM 只作「生成」(候选),判定仍走确定性(guard 预筛 / 行为效果 / 人审)。
|
|
199
|
+
|
|
200
|
+
const PROSE_SELECT_PROMPT_VERSION = '1.0.0'
|
|
201
|
+
|
|
202
|
+
function buildSelectSkillPrompt(signal: CrsiSignal, skillFiles: string[]): string {
|
|
203
|
+
return [
|
|
204
|
+
`你是 CRSI producer(producer-prose-select v${PROSE_SELECT_PROMPT_VERSION})。给定失败信号,从候选 skill 文件列表中选出最相关的一个,返回其文件路径(只返回路径,一行,不要其他文字)。`,
|
|
205
|
+
'',
|
|
206
|
+
'失败信号:',
|
|
207
|
+
`- category: ${signal.category}`,
|
|
208
|
+
`- title: ${signal.title}`,
|
|
209
|
+
signal.severity ? `- severity: ${signal.severity}` : '',
|
|
210
|
+
`- suggestion: ${signal.suggestion}`,
|
|
211
|
+
`- evidence: ${signal.evidence.join(' | ')}`,
|
|
212
|
+
'',
|
|
213
|
+
'候选 skill 文件:',
|
|
214
|
+
...skillFiles.map((f) => `- ${f}`),
|
|
215
|
+
]
|
|
216
|
+
.filter(Boolean)
|
|
217
|
+
.join('\n')
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
async function collectLlmText(llm: Llm, prompt: string): Promise<string> {
|
|
221
|
+
let text = ''
|
|
222
|
+
const req = {
|
|
223
|
+
model: 'prose',
|
|
224
|
+
messages: [{ role: 'user' as const, content: prompt }],
|
|
225
|
+
systemPrompt: '',
|
|
226
|
+
}
|
|
227
|
+
for await (const chunk of llm.chat(req)) {
|
|
228
|
+
if (chunk.type === 'text' && chunk.content) text += chunk.content
|
|
229
|
+
}
|
|
230
|
+
return text.trim()
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function extractFilePath(response: string, skillFiles: string[]): string | null {
|
|
234
|
+
for (const f of skillFiles) {
|
|
235
|
+
if (response.includes(f)) return f
|
|
236
|
+
}
|
|
237
|
+
return null
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
export async function selectTargetSkill(
|
|
241
|
+
signal: CrsiSignal,
|
|
242
|
+
llm: Llm,
|
|
243
|
+
skillFiles: string[],
|
|
244
|
+
): Promise<string | null> {
|
|
245
|
+
if (skillFiles.length === 0) return null
|
|
246
|
+
const prompt = buildSelectSkillPrompt(signal, skillFiles)
|
|
247
|
+
const response = await collectLlmText(llm, prompt)
|
|
248
|
+
if (!response) return null
|
|
249
|
+
return extractFilePath(response, skillFiles)
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
const PROSE_GENERATE_PROMPT_VERSION = '1.0.0'
|
|
253
|
+
|
|
254
|
+
function buildGenerateProsePrompt(
|
|
255
|
+
signal: CrsiSignal,
|
|
256
|
+
filePath: string,
|
|
257
|
+
originalContent: string,
|
|
258
|
+
): string {
|
|
259
|
+
return [
|
|
260
|
+
`你是 CRSI producer(producer-prose-generate v${PROSE_GENERATE_PROMPT_VERSION})。基于失败信号,改进目标 skill 的内容。`,
|
|
261
|
+
'',
|
|
262
|
+
'失败信号:',
|
|
263
|
+
`- category: ${signal.category}`,
|
|
264
|
+
`- title: ${signal.title}`,
|
|
265
|
+
`- suggestion: ${signal.suggestion}`,
|
|
266
|
+
`- evidence: ${signal.evidence.join(' | ')}`,
|
|
267
|
+
'',
|
|
268
|
+
`目标文件:${filePath}`,
|
|
269
|
+
'',
|
|
270
|
+
'当前内容:',
|
|
271
|
+
originalContent,
|
|
272
|
+
'',
|
|
273
|
+
'请返回改进后的完整 markdown(保持 YAML frontmatter 的 name/description 字段,正文针对失败信号做针对性改进)。只返回 markdown,不要额外说明。',
|
|
274
|
+
].join('\n')
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
function stripMarkdownFence(text: string): string {
|
|
278
|
+
const match = text.match(/^```(?:markdown|md)?\s*\n([\s\S]*?)\n```\s*$/)
|
|
279
|
+
return match ? match[1]! : text
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
export async function generateProseContent(
|
|
283
|
+
signal: CrsiSignal,
|
|
284
|
+
llm: Llm,
|
|
285
|
+
filePath: string,
|
|
286
|
+
originalContent: string,
|
|
287
|
+
): Promise<string | null> {
|
|
288
|
+
const prompt = buildGenerateProsePrompt(signal, filePath, originalContent)
|
|
289
|
+
const response = await collectLlmText(llm, prompt)
|
|
290
|
+
if (!response) return null
|
|
291
|
+
return stripMarkdownFence(response)
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
export interface ProseProposalResult {
|
|
295
|
+
filePath: string
|
|
296
|
+
newContent: string
|
|
297
|
+
originalContent: string
|
|
298
|
+
description: string
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
export async function produceProseProposal(
|
|
302
|
+
signal: CrsiSignal,
|
|
303
|
+
llm: Llm,
|
|
304
|
+
skillFiles: string[],
|
|
305
|
+
readSkill: (filePath: string) => string,
|
|
306
|
+
): Promise<ProseProposalResult | null> {
|
|
307
|
+
const filePath = await selectTargetSkill(signal, llm, skillFiles)
|
|
308
|
+
if (!filePath) return null
|
|
309
|
+
|
|
310
|
+
let originalContent: string
|
|
311
|
+
try {
|
|
312
|
+
originalContent = readSkill(filePath)
|
|
313
|
+
} catch {
|
|
314
|
+
return null
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
const newContent = await generateProseContent(signal, llm, filePath, originalContent)
|
|
318
|
+
if (!newContent) return null
|
|
319
|
+
|
|
320
|
+
return { filePath, newContent, originalContent, description: signal.title }
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
const SKILL_DIRS: Array<[string, string]> = [
|
|
324
|
+
['standard', '.SKILL.md'],
|
|
325
|
+
['mipham', '.mipham-skill.md'],
|
|
326
|
+
]
|
|
327
|
+
|
|
328
|
+
/** 收集仓库内所有 skill 文件(相对仓库根的路径),供 produceProseProposal 选目标。 */
|
|
329
|
+
export function collectSkillFiles(root: string): string[] {
|
|
330
|
+
const files: string[] = []
|
|
331
|
+
for (const [dir, ext] of SKILL_DIRS) {
|
|
332
|
+
let entries: string[] = []
|
|
333
|
+
try {
|
|
334
|
+
entries = readdirSync(join(root, 'apps', 'cli', 'skills', dir))
|
|
335
|
+
} catch {
|
|
336
|
+
continue
|
|
337
|
+
}
|
|
338
|
+
for (const entry of entries) {
|
|
339
|
+
if (entry.endsWith(ext)) files.push(`apps/cli/skills/${dir}/${entry}`)
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
return files
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// ── 幂等去重(prose ledger) ──
|
|
346
|
+
// 散文提议(块 1)的幂等:同一失败信号只生成一次提议。与 --rule 路径「目标文件内 id marker」去重不同,
|
|
347
|
+
// 散文改的是 skill 内容(非追加 marker),故用 ~/.mipham 下的 append-only ledger 记录「已提议的信号」。
|
|
348
|
+
|
|
349
|
+
/** 散文提议的稳定 id(同 category + 同 title → 同 id,同 managedRuleId 的 hash 语义)。 */
|
|
350
|
+
export function proseProposalId(signal: CrsiSignal): string {
|
|
351
|
+
return `prose-${signal.category}-${stableHash(signal.title)}`
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
/** ledger 里的一条散文提议记录。 */
|
|
355
|
+
export interface ProseProposalRecord {
|
|
356
|
+
id: string
|
|
357
|
+
filePath: string
|
|
358
|
+
timestamp: string
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
function proseLedgerFile(): string {
|
|
362
|
+
return join(homedir(), '.mipham', 'crsi', 'prose-proposals.jsonl')
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
/** 该信号是否已生成过散文提议。 */
|
|
366
|
+
export function hasProposedProse(id: string): boolean {
|
|
367
|
+
try {
|
|
368
|
+
if (!existsSync(proseLedgerFile())) return false
|
|
369
|
+
const lines = readFileSync(proseLedgerFile(), 'utf-8').trim().split('\n').filter(Boolean)
|
|
370
|
+
return lines.some((line) => {
|
|
371
|
+
try {
|
|
372
|
+
return (JSON.parse(line) as { id?: string }).id === id
|
|
373
|
+
} catch {
|
|
374
|
+
return false
|
|
375
|
+
}
|
|
376
|
+
})
|
|
377
|
+
} catch {
|
|
378
|
+
return false
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
/** 追加一条散文提议记录(append-only,非关键——失败不影响提议本身)。 */
|
|
383
|
+
export function appendProseProposal(record: ProseProposalRecord): void {
|
|
384
|
+
try {
|
|
385
|
+
mkdirSync(join(homedir(), '.mipham', 'crsi'), { recursive: true })
|
|
386
|
+
appendFileSync(proseLedgerFile(), JSON.stringify(record) + '\n', 'utf-8')
|
|
387
|
+
} catch {
|
|
388
|
+
// ledger 非关键,失败不影响提议本身
|
|
389
|
+
}
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/** 清空散文提议 ledger,返回清除的记录数(无文件时返回 0)。 */
|
|
393
|
+
export function clearProseProposals(): number {
|
|
394
|
+
try {
|
|
395
|
+
if (!existsSync(proseLedgerFile())) return 0
|
|
396
|
+
const lines = readFileSync(proseLedgerFile(), 'utf-8').trim().split('\n').filter(Boolean)
|
|
397
|
+
rmSync(proseLedgerFile(), { force: true })
|
|
398
|
+
return lines.length
|
|
399
|
+
} catch {
|
|
400
|
+
return 0
|
|
401
|
+
}
|
|
402
|
+
}
|
package/src/core/crsi-sandbox.ts
CHANGED
|
@@ -107,10 +107,15 @@ const PROTECTED_PATHS = [
|
|
|
107
107
|
'apps/cli/src/vajra/constitution.ts',
|
|
108
108
|
// eval harness
|
|
109
109
|
'apps/cli/test/',
|
|
110
|
+
'apps/cli/src/core/eval-harness.ts',
|
|
111
|
+
'apps/cli/src/core/behavior-tasks.ts',
|
|
112
|
+
'apps/cli/src/core/behavior-tasks.json',
|
|
110
113
|
// 改进机制自身
|
|
111
114
|
'apps/cli/src/agent/effectiveness-tracker.ts',
|
|
112
115
|
'apps/cli/src/core/meta-rule-engine.ts',
|
|
113
116
|
'apps/cli/src/core/crsi-sandbox.ts',
|
|
117
|
+
'apps/cli/src/core/crsi-producer.ts',
|
|
118
|
+
'apps/cli/src/core/proposal-guard.ts',
|
|
114
119
|
]
|
|
115
120
|
|
|
116
121
|
/** 是否命中只读边界。前缀匹配,目录条目以 `/` 结尾。 */
|
package/src/core/eval-harness.ts
CHANGED
|
@@ -22,6 +22,7 @@ import { RedTeam } from './red-team'
|
|
|
22
22
|
import { isProtectedPath } from './crsi-sandbox'
|
|
23
23
|
import { produceRuleProposal, MANAGED_RULES_FILE } from './crsi-producer'
|
|
24
24
|
import type { CrsiSignal } from './crsi-producer'
|
|
25
|
+
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
25
26
|
|
|
26
27
|
// ── Types ──
|
|
27
28
|
|
|
@@ -223,6 +224,12 @@ export function runEval(): EvalReport {
|
|
|
223
224
|
})
|
|
224
225
|
}
|
|
225
226
|
|
|
227
|
+
// ── 行为任务集(ground truth:约束行为效果,确定性无 LLM) ──
|
|
228
|
+
const behaviorTasks = loadBehaviorTasks()
|
|
229
|
+
for (const task of behaviorTasks) {
|
|
230
|
+
results.push(judgeBehaviorTask(task, ruleEngine))
|
|
231
|
+
}
|
|
232
|
+
|
|
226
233
|
const passed = results.filter((r) => r.passed).length
|
|
227
234
|
return {
|
|
228
235
|
total: results.length,
|
package/src/core/instructions.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { readFileSync, existsSync } from 'node:fs'
|
|
|
2
2
|
import { join } from 'node:path'
|
|
3
3
|
import { parse as parseYaml } from 'yaml'
|
|
4
4
|
import type { InstructionFile } from '../shared/index.ts'
|
|
5
|
+
import { COAUTHOR_TRAILER } from '../shared/index.ts'
|
|
5
6
|
|
|
6
7
|
interface FrontmatterResult {
|
|
7
8
|
data: Record<string, unknown>
|
|
@@ -163,6 +164,16 @@ its live CRSI / SIS / constitution state. Report the numbers you read
|
|
|
163
164
|
from it as live counts; if it shows a subsystem as 未初始化 (uninitialized),
|
|
164
165
|
say so explicitly instead of claiming it exists.`)
|
|
165
166
|
|
|
167
|
+
// AI 署名披露:提交时附带 Co-Authored-By 署名(与 Undercover 式隐瞒相反)
|
|
168
|
+
parts.push(`## Commit Attribution
|
|
169
|
+
|
|
170
|
+
When you create a git commit, always append this trailer on its own line
|
|
171
|
+
at the end of the commit message, disclosing AI involvement:
|
|
172
|
+
|
|
173
|
+
${COAUTHOR_TRAILER}
|
|
174
|
+
|
|
175
|
+
Never omit it or present the work as purely human-authored.`)
|
|
176
|
+
|
|
166
177
|
return parts.join('\n\n---\n\n')
|
|
167
178
|
}
|
|
168
179
|
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// apps/cli/src/core/proposal-guard.ts
|
|
2
|
+
// CRSI 提案预筛器(块 2 最小版):结构确定性预筛,零 LLM。
|
|
3
|
+
//
|
|
4
|
+
// 这是 CRSI 自改进「producer 改散文」闭环的第一道闸(三层验证的第①层):
|
|
5
|
+
// ① 结构确定性(本文件):受保护路径 + 自引用封闭 + 目标范围 + 结构不变量
|
|
6
|
+
// ② 行为效果(M3/A:LLM 生成行为 → 测试判定)—— 留作 seam,未实现
|
|
7
|
+
// ③ 人类审批 —— CrsiSandbox 之后
|
|
8
|
+
//
|
|
9
|
+
// 诚实边界:本预筛只能证「结构合法」,不能证「散文更好」——
|
|
10
|
+
// 后者是第②层(行为效果)的职责,需要 LLM 生成行为,属另一 A1 边界决策。
|
|
11
|
+
|
|
12
|
+
import { parse as parseYaml } from 'yaml'
|
|
13
|
+
import { isProtectedPath } from './crsi-sandbox'
|
|
14
|
+
import { MANAGED_RULES_FILE, MANAGED_RULE_MARKER } from './crsi-producer'
|
|
15
|
+
|
|
16
|
+
/** producer 产出的一条「改散文」提议(最小字段集,guard 阶段只消费这些)。 */
|
|
17
|
+
export interface ProducerProposal {
|
|
18
|
+
id: string
|
|
19
|
+
/** 仓库根相对路径 */
|
|
20
|
+
filePath: string
|
|
21
|
+
/** 最小版只放行两类目标,收窄散文风险面 */
|
|
22
|
+
kind: 'skill' | 'managed-rule'
|
|
23
|
+
/** 改后的完整文件内容 */
|
|
24
|
+
newContent: string
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export interface PrefilterVerdict {
|
|
28
|
+
pass: boolean
|
|
29
|
+
/** 拒绝理由;pass=true 时为空 */
|
|
30
|
+
reasons: string[]
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const SKILL_DIR = 'apps/cli/skills/'
|
|
34
|
+
|
|
35
|
+
function isSkillPath(filePath: string): boolean {
|
|
36
|
+
return (
|
|
37
|
+
filePath.startsWith(SKILL_DIR) &&
|
|
38
|
+
(filePath.endsWith('.SKILL.md') || filePath.endsWith('.mipham-skill.md'))
|
|
39
|
+
)
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* 更严、独立于 skills loader 的最小 frontmatter 解析。
|
|
44
|
+
*
|
|
45
|
+
* 与 loader 的宽松回退不同:frontmatter 缺失 / YAML 非法 / name 或 description
|
|
46
|
+
* 为空,都判失败(返回 null)——proposal 必须显式携带合法头部,不得依赖
|
|
47
|
+
* loader 的「文件名兜底 name」「空 description」等宽进逻辑。
|
|
48
|
+
*/
|
|
49
|
+
function parseStrictSkillFrontmatter(raw: string): { name: string; description: string } | null {
|
|
50
|
+
const match = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n/)
|
|
51
|
+
if (!match) return null
|
|
52
|
+
|
|
53
|
+
let data: unknown
|
|
54
|
+
try {
|
|
55
|
+
data = parseYaml(match[1] || '')
|
|
56
|
+
} catch {
|
|
57
|
+
return null
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
if (typeof data !== 'object' || data === null || Array.isArray(data)) return null
|
|
61
|
+
const obj = data as Record<string, unknown>
|
|
62
|
+
|
|
63
|
+
const name = typeof obj.name === 'string' ? obj.name.trim() : ''
|
|
64
|
+
const description = typeof obj.description === 'string' ? obj.description.trim() : ''
|
|
65
|
+
if (!name || !description) return null
|
|
66
|
+
return { name, description }
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export function prefilterProposal(p: ProducerProposal): PrefilterVerdict {
|
|
70
|
+
const reasons: string[] = []
|
|
71
|
+
|
|
72
|
+
// ① 受保护路径 + 自引用封闭(isProtectedPath 已含评估机制自身的 5 个补洞)
|
|
73
|
+
if (isProtectedPath(p.filePath)) {
|
|
74
|
+
reasons.push(`protected path: ${p.filePath} is read-only to the self-improvement loop`)
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// ② 目标范围白名单
|
|
78
|
+
const inScope =
|
|
79
|
+
(p.kind === 'skill' && isSkillPath(p.filePath)) ||
|
|
80
|
+
(p.kind === 'managed-rule' && p.filePath === MANAGED_RULES_FILE)
|
|
81
|
+
if (!inScope) {
|
|
82
|
+
reasons.push(`out of scope: ${p.filePath} is not an allowed ${p.kind} target`)
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// ③ 结构不变量
|
|
86
|
+
if (p.kind === 'skill') {
|
|
87
|
+
if (!parseStrictSkillFrontmatter(p.newContent)) {
|
|
88
|
+
reasons.push(
|
|
89
|
+
'skill frontmatter invalid: missing/illegal frontmatter or empty name/description',
|
|
90
|
+
)
|
|
91
|
+
}
|
|
92
|
+
} else if (!p.newContent.includes(MANAGED_RULE_MARKER)) {
|
|
93
|
+
reasons.push('managed rule missing the producer append marker')
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// ②(行为效果 seam):此处接入 M3/A 任务表现度量——用 LLM 生成行为 + 确定性
|
|
97
|
+
// 测试判定,过滤「行为效果退化」的提议。未实现,属 producer LLM 生成能力(块 1)。
|
|
98
|
+
|
|
99
|
+
return { pass: reasons.length === 0, reasons }
|
|
100
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"tasks": [
|
|
4
|
+
{
|
|
5
|
+
"id": "task-answer-fn",
|
|
6
|
+
"instruction": "用 Write 工具在 <taskDir>/solution.ts 写入一个 TypeScript 文件,导出 `export function answer(): number { return 42 }`。",
|
|
7
|
+
"groundTruth": {
|
|
8
|
+
"kind": "file-contains",
|
|
9
|
+
"file": "solution.ts",
|
|
10
|
+
"contains": ["export function answer", "42"]
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
]
|
|
14
|
+
}
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
// apps/cli/src/core/task-runner.ts
|
|
2
|
+
// CRSI 端到端任务运行器(C-MVP)——行为效果度量基建。
|
|
3
|
+
import { existsSync, readFileSync, mkdirSync, rmSync } from 'node:fs'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import tasksFile from './task-runner-tasks.json' with { type: 'json' }
|
|
6
|
+
import { QueryEngine } from './engine'
|
|
7
|
+
import { ContextManager } from './context'
|
|
8
|
+
import { PermissionSystem } from './permission'
|
|
9
|
+
import { ProviderRegistry } from '../providers/registry'
|
|
10
|
+
import type { Llm } from '../providers/llm'
|
|
11
|
+
import { createToolRegistry } from '../tools'
|
|
12
|
+
import type { PermissionLevel } from '../shared'
|
|
13
|
+
|
|
14
|
+
export type RunnerGroundTruth = { kind: 'file-contains'; file: string; contains: string[] }
|
|
15
|
+
|
|
16
|
+
export interface RunnerTask {
|
|
17
|
+
id: string
|
|
18
|
+
instruction: string
|
|
19
|
+
groundTruth: RunnerGroundTruth
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function loadRunnerTasks(): RunnerTask[] {
|
|
23
|
+
return tasksFile.tasks as unknown as RunnerTask[]
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function judgeTask(task: RunnerTask, taskDir: string): { passed: boolean; detail?: string } {
|
|
27
|
+
if (task.groundTruth.kind !== 'file-contains') {
|
|
28
|
+
return { passed: false, detail: `unsupported groundTruth kind: ${task.groundTruth.kind}` }
|
|
29
|
+
}
|
|
30
|
+
const filePath = join(taskDir, task.groundTruth.file)
|
|
31
|
+
if (!existsSync(filePath)) {
|
|
32
|
+
return { passed: false, detail: `file not found: ${task.groundTruth.file}` }
|
|
33
|
+
}
|
|
34
|
+
const content = readFileSync(filePath, 'utf-8')
|
|
35
|
+
for (const needle of task.groundTruth.contains) {
|
|
36
|
+
if (!content.includes(needle)) {
|
|
37
|
+
return { passed: false, detail: `missing substring: ${needle}` }
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return { passed: true }
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface TaskRunResult {
|
|
44
|
+
taskId: string
|
|
45
|
+
passed: boolean
|
|
46
|
+
detail?: string
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const TASK_DIR_PLACEHOLDER = '<taskDir>'
|
|
50
|
+
|
|
51
|
+
function buildEngine(llm: Llm, permission: PermissionLevel, systemPrompt?: string): QueryEngine {
|
|
52
|
+
const registry = new ProviderRegistry([], 'test', 'test-model')
|
|
53
|
+
// 注册一个永不 chat 的占位 provider——llm 被 setLlm 覆盖,但 process() 内部
|
|
54
|
+
// 多处调用 registry.getActive().config.id 记录 provider id,必须能取到。
|
|
55
|
+
registry.register('test', {
|
|
56
|
+
config: { id: 'test', name: 'Test', protocol: 'openai-compatible', apiKey: 'key', models: [] },
|
|
57
|
+
chat: async function* () {
|
|
58
|
+
yield { type: 'stop' }
|
|
59
|
+
},
|
|
60
|
+
listModels: async () => [],
|
|
61
|
+
healthCheck: async () => true,
|
|
62
|
+
})
|
|
63
|
+
const context = new ContextManager({ maxTokens: 100_000, compactionThreshold: 0.9 })
|
|
64
|
+
if (systemPrompt !== undefined) context.setSystemPrompt(systemPrompt)
|
|
65
|
+
const tools = createToolRegistry()
|
|
66
|
+
const engine = new QueryEngine(registry, context, tools, new PermissionSystem(permission))
|
|
67
|
+
engine.setLlm(llm)
|
|
68
|
+
return engine
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export async function runTask(
|
|
72
|
+
task: RunnerTask,
|
|
73
|
+
llm: Llm,
|
|
74
|
+
opts: { taskDir?: string; permission?: PermissionLevel; systemPrompt?: string } = {},
|
|
75
|
+
): Promise<TaskRunResult> {
|
|
76
|
+
const taskDir = opts.taskDir ?? join(process.cwd(), '.mipham', 'task-runner')
|
|
77
|
+
const permission = opts.permission ?? 'bypassPermissions'
|
|
78
|
+
|
|
79
|
+
rmSync(taskDir, { recursive: true, force: true })
|
|
80
|
+
mkdirSync(taskDir, { recursive: true })
|
|
81
|
+
|
|
82
|
+
const instruction = task.instruction.replaceAll(TASK_DIR_PLACEHOLDER, taskDir)
|
|
83
|
+
const engine = buildEngine(llm, permission, opts.systemPrompt)
|
|
84
|
+
|
|
85
|
+
for await (const _ of engine.process(instruction)) {
|
|
86
|
+
/* drain agentic loop */
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
const verdict = judgeTask(task, taskDir)
|
|
90
|
+
return { taskId: task.id, passed: verdict.passed, detail: verdict.detail }
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export interface TaskRunStats {
|
|
94
|
+
taskId: string
|
|
95
|
+
samples: number
|
|
96
|
+
passed: number
|
|
97
|
+
/** 0-1 */
|
|
98
|
+
passRate: number
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export async function runTaskN(
|
|
102
|
+
task: RunnerTask,
|
|
103
|
+
llm: Llm,
|
|
104
|
+
n: number,
|
|
105
|
+
opts: { taskDir?: string; permission?: PermissionLevel; systemPrompt?: string } = {},
|
|
106
|
+
): Promise<TaskRunStats> {
|
|
107
|
+
let passed = 0
|
|
108
|
+
for (let i = 0; i < n; i++) {
|
|
109
|
+
const result = await runTask(task, llm, opts)
|
|
110
|
+
if (result.passed) passed++
|
|
111
|
+
}
|
|
112
|
+
return { taskId: task.id, samples: n, passed, passRate: n > 0 ? passed / n : 0 }
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export interface RunComparison {
|
|
116
|
+
baseline: TaskRunStats
|
|
117
|
+
candidate: TaskRunStats
|
|
118
|
+
/** 弱判:candidate 不退化(不低于 baseline 且至少 1 次成功) */
|
|
119
|
+
notDegraded: boolean
|
|
120
|
+
/** candidate 严格更好(通过率更高) */
|
|
121
|
+
improved: boolean
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export function isNotDegraded(baseline: TaskRunStats, candidate: TaskRunStats): boolean {
|
|
125
|
+
return candidate.passRate >= baseline.passRate && candidate.passed >= 1
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
export function isImproved(baseline: TaskRunStats, candidate: TaskRunStats): boolean {
|
|
129
|
+
return candidate.passRate > baseline.passRate
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
export function compareRuns(baseline: TaskRunStats, candidate: TaskRunStats): RunComparison {
|
|
133
|
+
return {
|
|
134
|
+
baseline,
|
|
135
|
+
candidate,
|
|
136
|
+
notDegraded: isNotDegraded(baseline, candidate),
|
|
137
|
+
improved: isImproved(baseline, candidate),
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
export async function runBeforeAfter(
|
|
142
|
+
task: RunnerTask,
|
|
143
|
+
llm: Llm,
|
|
144
|
+
n: number,
|
|
145
|
+
opts: {
|
|
146
|
+
beforePrompt?: string
|
|
147
|
+
afterPrompt?: string
|
|
148
|
+
taskDir?: string
|
|
149
|
+
permission?: PermissionLevel
|
|
150
|
+
} = {},
|
|
151
|
+
): Promise<RunComparison> {
|
|
152
|
+
const baseline = await runTaskN(task, llm, n, {
|
|
153
|
+
taskDir: opts.taskDir,
|
|
154
|
+
permission: opts.permission,
|
|
155
|
+
systemPrompt: opts.beforePrompt,
|
|
156
|
+
})
|
|
157
|
+
const candidate = await runTaskN(task, llm, n, {
|
|
158
|
+
taskDir: opts.taskDir,
|
|
159
|
+
permission: opts.permission,
|
|
160
|
+
systemPrompt: opts.afterPrompt,
|
|
161
|
+
})
|
|
162
|
+
return compareRuns(baseline, candidate)
|
|
163
|
+
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// apps/cli/src/daemon/heartbeat.ts — 心跳式通知(保守版 KAIROS「订阅与推送」)
|
|
2
|
+
//
|
|
3
|
+
// 借鉴 Claude Code KAIROS 的「主动感知」思路,但严格约束为「只通知、不自主行动」:
|
|
4
|
+
// - 定时扫描 pending 的 goal(active)与 schedule(enabled)
|
|
5
|
+
// - 有 pending 时推送一条摘要(默认走 Feishu),无 pending 时静默
|
|
6
|
+
// - 绝不替用户执行任何动作——「主动感知可以,自主行动必须有闸门」(CRSI 受约束哲学)
|
|
7
|
+
//
|
|
8
|
+
// 纯函数(collectPendingItems / buildHeartbeatMessage / heartbeatTick)便于测试;
|
|
9
|
+
// startHeartbeat 只做 setInterval + unref 的薄接线。
|
|
10
|
+
|
|
11
|
+
import type { DaemonGoal, DaemonSchedule } from './types'
|
|
12
|
+
|
|
13
|
+
/** 默认心跳间隔:30 分钟(提醒型通知,不宜过频)。 */
|
|
14
|
+
export const DEFAULT_HEARTBEAT_INTERVAL_MS = 30 * 60_000
|
|
15
|
+
|
|
16
|
+
export interface PendingItems {
|
|
17
|
+
goalCount: number
|
|
18
|
+
scheduleCount: number
|
|
19
|
+
summaries: string[]
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** 收集待办项:active 的 goal + enabled 的 schedule。纯函数。 */
|
|
23
|
+
export function collectPendingItems(
|
|
24
|
+
goals: DaemonGoal[],
|
|
25
|
+
schedules: DaemonSchedule[],
|
|
26
|
+
): PendingItems {
|
|
27
|
+
const activeGoals = goals.filter((g) => g.status === 'active')
|
|
28
|
+
const enabledSchedules = schedules.filter((s) => s.enabled)
|
|
29
|
+
|
|
30
|
+
const summaries: string[] = []
|
|
31
|
+
for (const g of activeGoals) summaries.push(`🎯 ${g.description}`)
|
|
32
|
+
for (const s of enabledSchedules) summaries.push(`⏰ [${s.cronExpr}] ${s.prompt}`)
|
|
33
|
+
|
|
34
|
+
return {
|
|
35
|
+
goalCount: activeGoals.length,
|
|
36
|
+
scheduleCount: enabledSchedules.length,
|
|
37
|
+
summaries,
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const MAX_SUMMARY_ITEMS = 10
|
|
42
|
+
|
|
43
|
+
/** 把待办渲染成通知文案;无可待办时返回 null(不打扰)。纯函数。 */
|
|
44
|
+
export function buildHeartbeatMessage(pending: PendingItems): string | null {
|
|
45
|
+
if (pending.goalCount === 0 && pending.scheduleCount === 0) return null
|
|
46
|
+
|
|
47
|
+
const lines = [
|
|
48
|
+
`💓 Mipham 心跳提醒:${pending.goalCount} 个待办 goal,${pending.scheduleCount} 个定时任务`,
|
|
49
|
+
'',
|
|
50
|
+
...pending.summaries.slice(0, MAX_SUMMARY_ITEMS),
|
|
51
|
+
]
|
|
52
|
+
if (pending.summaries.length > MAX_SUMMARY_ITEMS) {
|
|
53
|
+
lines.push(`… 另外 ${pending.summaries.length - MAX_SUMMARY_ITEMS} 项`)
|
|
54
|
+
}
|
|
55
|
+
return lines.join('\n')
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export interface HeartbeatSource {
|
|
59
|
+
listGoals(): DaemonGoal[]
|
|
60
|
+
listSchedules(): DaemonSchedule[]
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** 单次心跳:收集待办,有则推送。抽出来便于直接测试。 */
|
|
64
|
+
export function heartbeatTick(source: HeartbeatSource, push: (message: string) => void): void {
|
|
65
|
+
const pending = collectPendingItems(source.listGoals(), source.listSchedules())
|
|
66
|
+
const message = buildHeartbeatMessage(pending)
|
|
67
|
+
if (message) push(message)
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export interface HeartbeatDeps {
|
|
71
|
+
source: HeartbeatSource
|
|
72
|
+
push: (message: string) => void
|
|
73
|
+
intervalMs?: number
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** 启动心跳定时器(unref 不阻止进程退出),返回 stop 函数。 */
|
|
77
|
+
export function startHeartbeat(deps: HeartbeatDeps): () => void {
|
|
78
|
+
const intervalMs = deps.intervalMs ?? DEFAULT_HEARTBEAT_INTERVAL_MS
|
|
79
|
+
const id = setInterval(() => heartbeatTick(deps.source, deps.push), intervalMs)
|
|
80
|
+
// Bun 的 Timer 有 unref;假定时器/其它运行时不保证——用可选调用容错。
|
|
81
|
+
;(id as { unref?: () => void }).unref?.()
|
|
82
|
+
return () => clearInterval(id)
|
|
83
|
+
}
|