@miphamai/cli 0.62.0 → 0.64.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/standard/safe-coding.SKILL.md +9 -0
- package/src/agent/effectiveness-tracker.ts +2 -6
- package/src/agent/message-bus.ts +10 -0
- package/src/agent/recoverable-failure.ts +16 -10
- package/src/agent/sub-agent.ts +15 -3
- package/src/config/defaults.ts +4 -0
- package/src/core/autocomplete.ts +64 -0
- package/src/core/crsi-modify.ts +13 -11
- package/src/core/crsi-producer.ts +128 -2
- package/src/core/crsi-sandbox.ts +83 -25
- package/src/core/engine.ts +30 -2
- package/src/core/eval-harness.ts +23 -8
- package/src/core/hooks-executor.ts +2 -1
- package/src/core/improvement-track.ts +144 -0
- package/src/core/reward-fn.ts +44 -0
- package/src/core/session-name.ts +12 -0
- package/src/core/task-performance-tasks.json +42 -0
- package/src/core/task-performance.ts +244 -0
- package/src/i18n-core/locales/en-US.json +3 -1
- package/src/i18n-core/locales/zh-CN.json +3 -1
- package/src/index.tsx +20 -8
- package/src/mcp/connect-failures.ts +15 -0
- package/src/plugin/claude-plugin.ts +1 -1
- package/src/plugin/plugin-validator.ts +1 -1
- package/src/shared/package-info.ts +1 -1
- package/src/shared/types.ts +7 -0
- package/src/shared/update.ts +41 -0
- package/src/skills/bundled-skills.ts +1 -0
- package/src/skills/loader.ts +1 -1
- package/src/ui/agent-footer.tsx +5 -1
- package/src/ui/app.tsx +71 -3
- package/src/ui/commands.ts +162 -8
- package/src/ui/input.tsx +71 -0
package/package.json
CHANGED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: safe-coding
|
|
3
|
+
description: Safe coding rules for code generation — validate external/user input before use and throw RangeError on invalid input
|
|
4
|
+
version: 1.0.0
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Safe Coding
|
|
8
|
+
|
|
9
|
+
处理外部/用户输入前必须校验:`null`、`undefined`、空字符串、格式非法时,抛出 `RangeError`,消息为 `'invalid input'`。
|
|
@@ -47,11 +47,7 @@ export class EffectivenessTracker {
|
|
|
47
47
|
if (eff) eff.decisionId = decisionId
|
|
48
48
|
}
|
|
49
49
|
|
|
50
|
-
recordApplication(
|
|
51
|
-
ruleId: string,
|
|
52
|
-
success: boolean,
|
|
53
|
-
opts?: { toolName?: string; error?: string },
|
|
54
|
-
): void {
|
|
50
|
+
recordApplication(ruleId: string, success: boolean, opts?: { error?: string }): void {
|
|
55
51
|
let eff = this.data.get(ruleId)
|
|
56
52
|
if (!eff) {
|
|
57
53
|
eff = {
|
|
@@ -69,7 +65,7 @@ export class EffectivenessTracker {
|
|
|
69
65
|
}
|
|
70
66
|
|
|
71
67
|
// 可恢复/环境性失败:仍记录,但不进成功率分母(防误降级/误禁用能用的规则)。
|
|
72
|
-
if (!success && opts?.error && isRecoverableToolFailure(opts.
|
|
68
|
+
if (!success && opts?.error && isRecoverableToolFailure(opts.error)) {
|
|
73
69
|
eff.recoverableCount = (eff.recoverableCount ?? 0) + 1
|
|
74
70
|
return
|
|
75
71
|
}
|
package/src/agent/message-bus.ts
CHANGED
|
@@ -24,6 +24,16 @@ export interface AgentMessage {
|
|
|
24
24
|
type: AgentMessageType
|
|
25
25
|
}
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Format an inbound message for injection into the conversation as a
|
|
29
|
+
* user-role notice. Collapsed to one line (`Message from @<sender>: <summary>`)
|
|
30
|
+
* so cross-session peer messages don't bloat the context; the full body stays
|
|
31
|
+
* in the bus for Ctrl+O expansion.
|
|
32
|
+
*/
|
|
33
|
+
export function formatInboundMessage(msg: AgentMessage): string {
|
|
34
|
+
return `Message from @${msg.from}: ${msg.summary}`
|
|
35
|
+
}
|
|
36
|
+
|
|
27
37
|
export class AgentMessageBus {
|
|
28
38
|
private messages: AgentMessage[] = []
|
|
29
39
|
private idCounter = 0
|
|
@@ -1,21 +1,27 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* 判「可恢复/环境性」失败 vs 真缺陷(borrow
|
|
2
|
+
* 判「可恢复/环境性」失败 vs 真缺陷(borrow openprabs is_recoverable_tool_failure 思想)。
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
4
|
+
* 可恢复失败(网络超时、连接拒绝、DNS 解析失败、环境权限)不该进成功率分母,否则会把
|
|
5
|
+
* 成功率拉低、诱导 EffectivenessTracker 禁用能用的规则(openprabs #236 的坑:stale-hash
|
|
6
|
+
* 重试被误判成工具缺陷,把能用的 hashline_edit 禁了)。
|
|
7
7
|
*
|
|
8
|
-
*
|
|
8
|
+
* 刻意排除 ENOENT(no such file or directory):缺文件/路径错更可能是规则本身指错了
|
|
9
|
+
* 路径(幻觉),而非环境瞬态,须计入分母(2026-08-27 review M1)。
|
|
10
|
+
*
|
|
11
|
+
* 不区分工具:这些错误模式(超时/连接/DNS/权限)本身即工具无关的环境信号。旧的
|
|
12
|
+
* `toolName !== 'Bash'` 守卫漏掉 Write/Grep 规则(2026-08-27 review M2),已移除。
|
|
13
|
+
*
|
|
14
|
+
* fail-closed:无 error 一律按真失败算(证不了可恢复就不排除)。
|
|
9
15
|
*/
|
|
10
16
|
const RECOVERABLE_PATTERNS: RegExp[] = [
|
|
11
|
-
/\btimed?\s?out\b/i, // timeout / timed out(网络慢 /
|
|
12
|
-
/(ECONNREFUSED|connection refused|not connected|network unreachable)/i, // 服务未起
|
|
13
|
-
/(
|
|
17
|
+
/\btimed?\s?out\b/i, // timeout / timed out / timedout(网络慢 / 服务慢)
|
|
18
|
+
/(ECONNREFUSED|ECONNRESET|ETIMEDOUT|connection refused|not connected|network unreachable)/i, // 服务未起 / 网络超时
|
|
19
|
+
/(could not resolve host|name or service not known|getaddrinfo|ENOTFOUND)/i, // DNS 解析失败
|
|
14
20
|
/(permission denied|EACCES)/i, // 环境权限
|
|
15
21
|
]
|
|
16
22
|
|
|
17
|
-
export function isRecoverableToolFailure(
|
|
18
|
-
if (!error
|
|
23
|
+
export function isRecoverableToolFailure(error?: string): boolean {
|
|
24
|
+
if (!error) return false
|
|
19
25
|
const e = error.toLowerCase()
|
|
20
26
|
return RECOVERABLE_PATTERNS.some((re) => re.test(e))
|
|
21
27
|
}
|
package/src/agent/sub-agent.ts
CHANGED
|
@@ -319,6 +319,7 @@ export class SubAgent {
|
|
|
319
319
|
|
|
320
320
|
const chunks: string[] = []
|
|
321
321
|
const MAX_TOOL_TURNS = options.maxTurns || 5
|
|
322
|
+
let hitMaxTurns = false
|
|
322
323
|
|
|
323
324
|
let currentMessages = messages
|
|
324
325
|
let currentSystemPrompt = systemPrompt
|
|
@@ -360,7 +361,7 @@ export class SubAgent {
|
|
|
360
361
|
})
|
|
361
362
|
}
|
|
362
363
|
if (chunk.type === 'error') {
|
|
363
|
-
throw new Error(`Sub-agent execution failed: ${chunk.error}`)
|
|
364
|
+
throw new Error(`Sub-agent execution failed (model ${finalModel}): ${chunk.error}`)
|
|
364
365
|
}
|
|
365
366
|
if (chunk.type === 'stop') {
|
|
366
367
|
break
|
|
@@ -501,6 +502,9 @@ export class SubAgent {
|
|
|
501
502
|
|
|
502
503
|
// Don't send system prompt on subsequent turns
|
|
503
504
|
currentSystemPrompt = ''
|
|
505
|
+
|
|
506
|
+
// On the final permitted turn we still used tools → hit the cap.
|
|
507
|
+
if (turn === MAX_TOOL_TURNS - 1) hitMaxTurns = true
|
|
504
508
|
}
|
|
505
509
|
} catch (err) {
|
|
506
510
|
if (err instanceof DOMException && err.name === 'AbortError') {
|
|
@@ -509,10 +513,18 @@ export class SubAgent {
|
|
|
509
513
|
if (err instanceof Error && err.message.startsWith('Sub-agent')) {
|
|
510
514
|
throw err
|
|
511
515
|
}
|
|
512
|
-
throw new Error(`Sub-agent execution failed: ${String(err)}`)
|
|
516
|
+
throw new Error(`Sub-agent execution failed (model ${finalModel}): ${String(err)}`)
|
|
513
517
|
}
|
|
514
518
|
|
|
515
|
-
|
|
519
|
+
const result = chunks.join('')
|
|
520
|
+
if (hitMaxTurns) {
|
|
521
|
+
return (
|
|
522
|
+
`[partial result — sub-agent reached its ${MAX_TOOL_TURNS}-turn limit; ` +
|
|
523
|
+
'task may be incomplete. Use SendMessage to continue this sub-agent.]\n\n' +
|
|
524
|
+
result
|
|
525
|
+
)
|
|
526
|
+
}
|
|
527
|
+
return result
|
|
516
528
|
}
|
|
517
529
|
|
|
518
530
|
/**
|
package/src/config/defaults.ts
CHANGED
|
@@ -16,6 +16,10 @@ export const DEFAULT_CONFIG: MiphamConfig = {
|
|
|
16
16
|
showThinking: 'off',
|
|
17
17
|
showSchedulingNotices: false,
|
|
18
18
|
showCommandPicker: false,
|
|
19
|
+
autocomplete: {
|
|
20
|
+
enabled: true,
|
|
21
|
+
debounceMs: 400,
|
|
22
|
+
},
|
|
19
23
|
// Org 级权限限制(可选):forbiddenModes 禁指定模式 / maxAllowedMode 封顶层级;
|
|
20
24
|
// 请求被禁模式时 fail-closed 降级(如 forbiddenModes:['bypassPermissions'])。
|
|
21
25
|
// permissionRestrictions: { forbiddenModes: ['bypassPermissions'] },
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { ChatRequest } from '../providers/registry'
|
|
2
|
+
import type { Llm } from '../providers/llm'
|
|
3
|
+
|
|
4
|
+
export const AUTOCOMPLETE_SYSTEM_PROMPT =
|
|
5
|
+
'你是续写助手。只续写用户正在输入的这条消息,只返回续写部分(不要重复已输入的文字、不要解释、不要换行)。'
|
|
6
|
+
|
|
7
|
+
/** 带上最近几条对话(含待续写输入),供续写贴合上下文。 */
|
|
8
|
+
export const AUTOCOMPLETE_MAX_CONTEXT = 6
|
|
9
|
+
|
|
10
|
+
export interface RecentMessage {
|
|
11
|
+
role: 'user' | 'assistant'
|
|
12
|
+
content: string
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/** 拼续写请求:systemPrompt + 最近 N 条 + 当前输入作为待续写消息。 */
|
|
16
|
+
export function buildAutocompleteRequest(recent: RecentMessage[], input: string): ChatRequest {
|
|
17
|
+
return {
|
|
18
|
+
model: '', // falsy → registry 回退 active model
|
|
19
|
+
messages: [...recent.slice(-AUTOCOMPLETE_MAX_CONTEXT), { role: 'user', content: input }],
|
|
20
|
+
systemPrompt: AUTOCOMPLETE_SYSTEM_PROMPT,
|
|
21
|
+
temperature: 0,
|
|
22
|
+
maxTokens: 64,
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** 剥掉 LLM 可能重复的 input 前缀,返回纯续写 suffix;空/无效 → null。 */
|
|
27
|
+
export function extractCompletion(response: string, input: string): string | null {
|
|
28
|
+
let completion = response.trim()
|
|
29
|
+
if (!completion) return null
|
|
30
|
+
const normInput = input.trim()
|
|
31
|
+
if (normInput && completion.startsWith(normInput)) {
|
|
32
|
+
completion = completion.slice(normInput.length).trimStart()
|
|
33
|
+
}
|
|
34
|
+
return completion || null
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 触发 guard:非空、非 `/`·`@` 开头、非 loading、无活跃 picker。 */
|
|
38
|
+
export function shouldAutocomplete(
|
|
39
|
+
value: string,
|
|
40
|
+
isLoading: boolean,
|
|
41
|
+
pickerActive: boolean,
|
|
42
|
+
): boolean {
|
|
43
|
+
if (!value.trim()) return false
|
|
44
|
+
if (value.startsWith('/') || value.startsWith('@')) return false
|
|
45
|
+
if (isLoading) return false
|
|
46
|
+
if (pickerActive) return false
|
|
47
|
+
return true
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** 异步取建议:llm.chat → 竞态检查(isStale)→ 剥前缀。stale / 空 → null。 */
|
|
51
|
+
export async function requestSuggestion(
|
|
52
|
+
llm: Llm,
|
|
53
|
+
recent: RecentMessage[],
|
|
54
|
+
input: string,
|
|
55
|
+
isStale: () => boolean,
|
|
56
|
+
): Promise<string | null> {
|
|
57
|
+
const req = buildAutocompleteRequest(recent, input)
|
|
58
|
+
let text = ''
|
|
59
|
+
for await (const chunk of llm.chat(req)) {
|
|
60
|
+
if (chunk.type === 'text' && chunk.content) text += chunk.content
|
|
61
|
+
}
|
|
62
|
+
if (isStale()) return null
|
|
63
|
+
return extractCompletion(text, input)
|
|
64
|
+
}
|
package/src/core/crsi-modify.ts
CHANGED
|
@@ -15,7 +15,8 @@
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
16
|
import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
|
|
17
17
|
import type { CrsiModificationResult } from './crsi-sandbox'
|
|
18
|
-
import {
|
|
18
|
+
import { appendEvalScore, getLastEvalScore } from './eval-harness'
|
|
19
|
+
import { mechanismSentinel, type RewardFn } from './reward-fn'
|
|
19
20
|
|
|
20
21
|
export interface CrsiProposal {
|
|
21
22
|
/** 人类可读的改动说明 */
|
|
@@ -45,10 +46,11 @@ let pendingSandbox: CrsiSandbox | null = null
|
|
|
45
46
|
* 编排完整 5 阶段:createWorktree → applyModification → runTests →(失败自动
|
|
46
47
|
* rollback)→ getDiff。测试通过后暂存为 pending,返回 diff 供人类审阅。
|
|
47
48
|
*/
|
|
48
|
-
export function runCrsiModification(
|
|
49
|
+
export async function runCrsiModification(
|
|
49
50
|
proposal: CrsiProposal,
|
|
50
51
|
sandbox: CrsiSandbox = new CrsiSandbox(),
|
|
51
|
-
|
|
52
|
+
opts?: { rewardFn?: RewardFn },
|
|
53
|
+
): Promise<CrsiModificationResult> {
|
|
52
54
|
// 完整覆盖闸:自修改必须声明 blastRadius,否则 fail-closed(在 worktree 之前,零副作用)。
|
|
53
55
|
const blastRadiusError = validateBlastRadius(proposal)
|
|
54
56
|
if (blastRadiusError) {
|
|
@@ -96,18 +98,18 @@ export function runCrsiModification(
|
|
|
96
98
|
return applied
|
|
97
99
|
}
|
|
98
100
|
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
|
|
102
|
-
const
|
|
103
|
-
const last = getLastEvalScore()
|
|
104
|
-
if (last !== null &&
|
|
101
|
+
// Reward gate:奖励分数不得低于上次记录(防跨合并退化)。
|
|
102
|
+
// 默认机制哨兵;可插拔——调用方传 opts.rewardFn 换用其他奖励源(如任务表现)。
|
|
103
|
+
const rewardFn = opts?.rewardFn ?? mechanismSentinel()
|
|
104
|
+
const report = await rewardFn.evaluate()
|
|
105
|
+
const last = getLastEvalScore(rewardFn.name)
|
|
106
|
+
if (last !== null && report.score < last) {
|
|
105
107
|
sandbox.rollback()
|
|
106
108
|
applied.phase = 'failed'
|
|
107
|
-
applied.error = `
|
|
109
|
+
applied.error = `Reward regression (${rewardFn.name}): score ${report.score} < last ${last}`
|
|
108
110
|
return applied
|
|
109
111
|
}
|
|
110
|
-
appendEvalScore(
|
|
112
|
+
appendEvalScore(rewardFn.name, report)
|
|
111
113
|
|
|
112
114
|
applied.phase = 'passed'
|
|
113
115
|
applied.diff = sandbox.getDiff()
|
|
@@ -67,14 +67,18 @@ export function selectCrsiSignal(
|
|
|
67
67
|
}
|
|
68
68
|
|
|
69
69
|
/** 模板化地把信号渲染成一段教训 markdown(不动 LLM)。 */
|
|
70
|
-
export function buildLessonContent(
|
|
70
|
+
export function buildLessonContent(
|
|
71
|
+
signal: CrsiSignal,
|
|
72
|
+
timestamp: string,
|
|
73
|
+
source = 'CRSI producer (autoApplicable)',
|
|
74
|
+
): string {
|
|
71
75
|
const lines: string[] = [
|
|
72
76
|
`## ${signal.category}: ${signal.title}`,
|
|
73
77
|
'',
|
|
74
78
|
`- 建议: ${signal.suggestion}`,
|
|
75
79
|
]
|
|
76
80
|
if (signal.severity) lines.push(`- 严重度: ${signal.severity}`)
|
|
77
|
-
lines.push(`- 生成时间: ${timestamp}`,
|
|
81
|
+
lines.push(`- 生成时间: ${timestamp}`, `- 来源: ${source}`, '', '### 证据')
|
|
78
82
|
for (const e of signal.evidence) lines.push(`- ${e}`)
|
|
79
83
|
lines.push('')
|
|
80
84
|
return lines.join('\n')
|
|
@@ -455,3 +459,125 @@ export function clearProseProposals(): number {
|
|
|
455
459
|
return 0
|
|
456
460
|
}
|
|
457
461
|
}
|
|
462
|
+
|
|
463
|
+
// ── Producer Crossover(第 4 原子算子):合并两条重叠教训 ──
|
|
464
|
+
// LLM 只生成(选对 + 合并版),判定全走确定性 guard + 沙箱 gate。A1 不破:无 LLM 自评。
|
|
465
|
+
|
|
466
|
+
const CROSSOVER_PROMPT_VERSION = '1.0.0'
|
|
467
|
+
|
|
468
|
+
function buildCrossoverPrompt(currentLessons: string): string {
|
|
469
|
+
return [
|
|
470
|
+
`你是 CRSI producer(producer-crossover v${CROSSOVER_PROMPT_VERSION})。给定当前教训文件,找出两条主题重叠、可合并的教训,生成一条综合教训。`,
|
|
471
|
+
'',
|
|
472
|
+
'当前教训文件:',
|
|
473
|
+
currentLessons,
|
|
474
|
+
'',
|
|
475
|
+
'要求:',
|
|
476
|
+
'1. 找两条「主题重叠」的教训(例如都讲「读码优先」、都讲「隔离」),不要选主题无关的两条。',
|
|
477
|
+
'2. titleA / titleB 是 `## ` 之后、标题的完整文本(含 category 前缀,逐字复制,不要改写;**不含** `## ` 前缀本身)。',
|
|
478
|
+
'3. merged 是合并后的综合教训:category 沿用其中一个、title 概括两者、suggestion 综合两条的核心建议、evidence 综合两条的证据要点。',
|
|
479
|
+
'4. 只返回裸 JSON(不要 markdown 围栏、不要其他文字),格式:',
|
|
480
|
+
'{"titleA":"<完整 ## 行1>","titleB":"<完整 ## 行2>","merged":{"category":"...","title":"...","suggestion":"...","evidence":["...","..."]}}',
|
|
481
|
+
].join('\n')
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/** 剥 ```json 围栏(LLM 可能加)。 */
|
|
485
|
+
function stripJsonFence(text: string): string {
|
|
486
|
+
const match = text.match(/^```(?:json)?\s*\n([\s\S]*?)\n```\s*$/)
|
|
487
|
+
return match ? match[1]! : text
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
/** Crossover 结果:两条教训的完整 ## 行 + 合并版。 */
|
|
491
|
+
export interface CrossoverResult {
|
|
492
|
+
titleA: string
|
|
493
|
+
titleB: string
|
|
494
|
+
merged: CrsiSignal
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
/** 解析 crossover 结果;非法 / 字段缺失 → null。 */
|
|
498
|
+
export function parseCrossoverResult(text: string): CrossoverResult | null {
|
|
499
|
+
try {
|
|
500
|
+
const obj = JSON.parse(stripJsonFence(text))
|
|
501
|
+
if (typeof obj.titleA !== 'string' || typeof obj.titleB !== 'string') return null
|
|
502
|
+
if (
|
|
503
|
+
!obj.merged ||
|
|
504
|
+
typeof obj.merged.category !== 'string' ||
|
|
505
|
+
typeof obj.merged.title !== 'string' ||
|
|
506
|
+
typeof obj.merged.suggestion !== 'string'
|
|
507
|
+
)
|
|
508
|
+
return null
|
|
509
|
+
const evidence = Array.isArray(obj.merged.evidence)
|
|
510
|
+
? obj.merged.evidence.filter((e: unknown) => typeof e === 'string')
|
|
511
|
+
: []
|
|
512
|
+
return {
|
|
513
|
+
titleA: obj.titleA,
|
|
514
|
+
titleB: obj.titleB,
|
|
515
|
+
merged: {
|
|
516
|
+
category: obj.merged.category,
|
|
517
|
+
title: obj.merged.title,
|
|
518
|
+
suggestion: obj.merged.suggestion,
|
|
519
|
+
evidence,
|
|
520
|
+
},
|
|
521
|
+
}
|
|
522
|
+
} catch {
|
|
523
|
+
return null
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
/** 从教训文件移除若干 `## ` 段(header 须是 `## ` 行完整文本)。preamble 与其余教训不动。 */
|
|
528
|
+
export function removeLessonSections(content: string, headers: string[]): string {
|
|
529
|
+
const lines = content.split('\n')
|
|
530
|
+
const out: string[] = []
|
|
531
|
+
let skipping = false
|
|
532
|
+
for (const line of lines) {
|
|
533
|
+
if (line.startsWith('## ')) {
|
|
534
|
+
skipping = headers.includes(line.trim())
|
|
535
|
+
if (skipping) continue
|
|
536
|
+
}
|
|
537
|
+
if (skipping) continue
|
|
538
|
+
out.push(line)
|
|
539
|
+
}
|
|
540
|
+
return out.join('\n')
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
/**
|
|
544
|
+
* Crossover:合并两条重叠教训 → 「删二增一」的教训文件变更候选。
|
|
545
|
+
* LLM 只生成(选对 + 合并版),guard 校验所选教训真实存在(fail-closed 防幻觉)。
|
|
546
|
+
*/
|
|
547
|
+
export async function produceCrossoverProposal(
|
|
548
|
+
llm: Llm,
|
|
549
|
+
currentLessons: string,
|
|
550
|
+
timestamp: string,
|
|
551
|
+
): Promise<{
|
|
552
|
+
description: string
|
|
553
|
+
filePath: string
|
|
554
|
+
newContent: string
|
|
555
|
+
originalContent: string
|
|
556
|
+
blastRadius: string[]
|
|
557
|
+
} | null> {
|
|
558
|
+
const response = await collectLlmText(llm, buildCrossoverPrompt(currentLessons))
|
|
559
|
+
if (!response) return null
|
|
560
|
+
|
|
561
|
+
const parsed = parseCrossoverResult(response)
|
|
562
|
+
if (!parsed) return null
|
|
563
|
+
|
|
564
|
+
const headerA = `## ${parsed.titleA}`
|
|
565
|
+
const headerB = `## ${parsed.titleB}`
|
|
566
|
+
if (parsed.titleA === parsed.titleB) return null
|
|
567
|
+
// 精确行匹配(与 removeLessonSections 同语义):子串 includes 会让「截断标题」漏过 guard、
|
|
568
|
+
// 却因 removeLessonSections 精确匹配删不掉 → 假「删二增一」实为「增一」。fail-closed 用精确匹配。
|
|
569
|
+
const lessonLines = currentLessons.split('\n').map((l) => l.trim())
|
|
570
|
+
if (!lessonLines.includes(headerA) || !lessonLines.includes(headerB)) return null
|
|
571
|
+
|
|
572
|
+
const withoutTwo = removeLessonSections(currentLessons, [headerA, headerB])
|
|
573
|
+
const mergedSection = buildLessonContent(parsed.merged, timestamp, 'CRSI producer (crossover)')
|
|
574
|
+
const newContent = `${withoutTwo.trimEnd()}\n\n${mergedSection}\n`
|
|
575
|
+
|
|
576
|
+
return {
|
|
577
|
+
description: `CRSI crossover: ${parsed.titleA} + ${parsed.titleB}`,
|
|
578
|
+
filePath: LESSONS_FILE,
|
|
579
|
+
newContent,
|
|
580
|
+
originalContent: currentLessons,
|
|
581
|
+
blastRadius: [LESSONS_FILE],
|
|
582
|
+
}
|
|
583
|
+
}
|
package/src/core/crsi-sandbox.ts
CHANGED
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
|
|
17
17
|
import { execSync } from 'node:child_process'
|
|
18
18
|
import { mkdirSync, rmSync, existsSync, writeFileSync, readFileSync, readdirSync } from 'node:fs'
|
|
19
|
-
import { join, resolve, sep } from 'node:path'
|
|
19
|
+
import { join, resolve, sep, posix } from 'node:path'
|
|
20
20
|
import { tmpdir, homedir } from 'node:os'
|
|
21
21
|
import { randomUUID } from 'node:crypto'
|
|
22
22
|
|
|
@@ -88,33 +88,77 @@ const TEST_TIMEOUT_MS = 120_000 // 2 minutes
|
|
|
88
88
|
const REPORT_DIR = join(homedir(), '.mipham', 'crsi-sandbox')
|
|
89
89
|
|
|
90
90
|
/**
|
|
91
|
-
*
|
|
91
|
+
* 自改进的「不可变基础」(immutable base)——按语义角色三类。
|
|
92
|
+
* 自改进循环可以改 skill/workflow/prompt/memory/教训/managed-rules,
|
|
93
|
+
* 但绝不能改以下三者,否则会削弱 grader 或安全边界:
|
|
94
|
+
* constitution —— 宪法/对齐:改掉 = 价值漂移优化掉安全边界
|
|
95
|
+
* evaluator —— 评估器/grader:改掉 = 改掉自己的评分标准(Goodhart 元劫持)
|
|
96
|
+
* selfImprovement —— 改进机制自身:改掉 = 递归改掉评估器/安全
|
|
92
97
|
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
98
|
+
* 注意:fail-closed——宁可多拦,不可漏拦。新增机制文件必须加进对应类别,
|
|
99
|
+
* 否则 eval harness 的 protection-completeness 契约会 fail。
|
|
100
|
+
*/
|
|
101
|
+
export const PROTECTED_ROLES = {
|
|
102
|
+
constitution: [
|
|
103
|
+
'apps/cli/src/core/alignment-vocabulary.json',
|
|
104
|
+
'apps/cli/src/core/constitution-loader.ts',
|
|
105
|
+
'apps/cli/src/core/constitution-seam.ts',
|
|
106
|
+
'apps/cli/src/vajra/constitution.ts',
|
|
107
|
+
],
|
|
108
|
+
evaluator: [
|
|
109
|
+
'apps/cli/test/',
|
|
110
|
+
'apps/cli/src/core/eval-harness.ts',
|
|
111
|
+
'apps/cli/src/core/behavior-tasks.ts',
|
|
112
|
+
'apps/cli/src/core/behavior-tasks.json',
|
|
113
|
+
'apps/cli/src/core/task-performance.ts',
|
|
114
|
+
'apps/cli/src/core/task-performance-tasks.json',
|
|
115
|
+
'apps/cli/src/core/improvement-track.ts',
|
|
116
|
+
'apps/cli/src/core/reward-fn.ts',
|
|
117
|
+
],
|
|
118
|
+
selfImprovement: [
|
|
119
|
+
'apps/cli/src/agent/effectiveness-tracker.ts',
|
|
120
|
+
'apps/cli/src/agent/recoverable-failure.ts',
|
|
121
|
+
'apps/cli/src/agent/crsi-provenance-bridge.ts',
|
|
122
|
+
'apps/cli/src/agent/experience-rules.ts',
|
|
123
|
+
'apps/cli/src/agent/agent-experience.ts',
|
|
124
|
+
'apps/cli/src/core/meta-rule-engine.ts',
|
|
125
|
+
'apps/cli/src/core/crsi-sandbox.ts',
|
|
126
|
+
'apps/cli/src/core/crsi-producer.ts',
|
|
127
|
+
'apps/cli/src/core/proposal-guard.ts',
|
|
128
|
+
'apps/cli/src/core/crsi-modify.ts',
|
|
129
|
+
'apps/cli/src/core/rule-engine.ts',
|
|
130
|
+
'apps/cli/src/core/red-team.ts',
|
|
131
|
+
'apps/cli/src/core/error-signature-db.ts',
|
|
132
|
+
'apps/cli/src/core/preflight-checker.ts',
|
|
133
|
+
'apps/cli/src/core/permission-rules.ts',
|
|
134
|
+
'apps/cli/src/core/rules-loader.ts',
|
|
135
|
+
],
|
|
136
|
+
} as const
|
|
137
|
+
|
|
138
|
+
/** 扁平化(向后兼容:isProtectedPath 仍用前缀匹配,行为不变)。 */
|
|
139
|
+
export const PROTECTED_PATHS: string[] = Object.values(PROTECTED_ROLES).flat()
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* 完整性金丝雀:这些「评估器 + 核心机制」文件必须全在保护域。
|
|
143
|
+
* eval harness 的 protection-completeness 契约逐条断言 isProtectedPath。
|
|
144
|
+
* 与 PROTECTED_ROLES 同文件(单一维护点)——新增机制文件须两处一起加。
|
|
100
145
|
*/
|
|
101
|
-
const
|
|
102
|
-
// 宪法(对齐)
|
|
103
|
-
'apps/cli/src/core/alignment-vocabulary.json',
|
|
104
|
-
'apps/cli/src/core/constitution-loader.ts',
|
|
105
|
-
'apps/cli/src/core/constitution-seam.ts',
|
|
106
|
-
'apps/cli/src/vajra/constitution.ts',
|
|
107
|
-
// eval harness
|
|
108
|
-
'apps/cli/test/',
|
|
146
|
+
export const PROTECTED_CRITICAL_FILES: string[] = [
|
|
109
147
|
'apps/cli/src/core/eval-harness.ts',
|
|
110
148
|
'apps/cli/src/core/behavior-tasks.ts',
|
|
111
149
|
'apps/cli/src/core/behavior-tasks.json',
|
|
112
|
-
|
|
113
|
-
'apps/cli/src/
|
|
114
|
-
'apps/cli/src/core/
|
|
150
|
+
'apps/cli/src/core/task-performance.ts',
|
|
151
|
+
'apps/cli/src/core/task-performance-tasks.json',
|
|
152
|
+
'apps/cli/src/core/improvement-track.ts',
|
|
153
|
+
'apps/cli/src/core/reward-fn.ts',
|
|
115
154
|
'apps/cli/src/core/crsi-sandbox.ts',
|
|
116
155
|
'apps/cli/src/core/crsi-producer.ts',
|
|
117
|
-
'apps/cli/src/core/
|
|
156
|
+
'apps/cli/src/core/rule-engine.ts',
|
|
157
|
+
'apps/cli/src/core/red-team.ts',
|
|
158
|
+
'apps/cli/src/core/error-signature-db.ts',
|
|
159
|
+
'apps/cli/src/core/preflight-checker.ts',
|
|
160
|
+
'apps/cli/src/agent/recoverable-failure.ts',
|
|
161
|
+
'apps/cli/src/agent/crsi-provenance-bridge.ts',
|
|
118
162
|
]
|
|
119
163
|
|
|
120
164
|
/** 是否命中只读边界。前缀匹配,目录条目以 `/` 结尾。 */
|
|
@@ -130,13 +174,24 @@ export function isProtectedPath(filePath: string): boolean {
|
|
|
130
174
|
*
|
|
131
175
|
* 返回错误字符串(拒绝理由),合法时返回 null。
|
|
132
176
|
*/
|
|
133
|
-
export function validateBlastRadius(proposal: {
|
|
177
|
+
export function validateBlastRadius(proposal: {
|
|
178
|
+
filePath?: string
|
|
179
|
+
blastRadius?: string[]
|
|
180
|
+
}): string | null {
|
|
181
|
+
const filePath = proposal.filePath
|
|
134
182
|
if (!proposal.blastRadius || proposal.blastRadius.length === 0) {
|
|
135
183
|
return (
|
|
136
184
|
'blast radius 未声明:自修改必须枚举触及的全部代码路径。' +
|
|
137
185
|
'教训:两条渲染路径只接一条 = 局部正确全局遗漏。'
|
|
138
186
|
)
|
|
139
187
|
}
|
|
188
|
+
// 2026-08-27 review:blastRadius 不能只自证「非空」,必须覆盖被修改文件本身
|
|
189
|
+
// (前缀匹配,目录条目以 / 结尾),否则声明形同虚设。
|
|
190
|
+
if (filePath && !proposal.blastRadius.some((p) => filePath === p || filePath.startsWith(p))) {
|
|
191
|
+
return (
|
|
192
|
+
`blast radius 未覆盖目标文件 "${filePath}":` + '声明必须包含被修改文件本身(前缀匹配)。'
|
|
193
|
+
)
|
|
194
|
+
}
|
|
140
195
|
return null
|
|
141
196
|
}
|
|
142
197
|
|
|
@@ -225,7 +280,10 @@ export class CrsiSandbox {
|
|
|
225
280
|
}
|
|
226
281
|
|
|
227
282
|
const worktreeRoot = resolve(this.worktreePath)
|
|
228
|
-
|
|
283
|
+
// Normalize "./" and "subdir/../" so the protected-path guard can't be evaded
|
|
284
|
+
// by a path that resolves back inside the worktree (raw prefix match would miss it).
|
|
285
|
+
const normalizedFilePath = posix.normalize(mod.filePath)
|
|
286
|
+
const targetPath = resolve(this.worktreePath, normalizedFilePath)
|
|
229
287
|
|
|
230
288
|
// Path traversal guard: reject any target that resolves outside the sandbox worktree
|
|
231
289
|
if (targetPath !== worktreeRoot && !targetPath.startsWith(worktreeRoot + sep)) {
|
|
@@ -237,7 +295,7 @@ export class CrsiSandbox {
|
|
|
237
295
|
|
|
238
296
|
// Protected-path guard: the self-improvement loop must not modify the
|
|
239
297
|
// constitution, eval harness, or improvement machinery itself.
|
|
240
|
-
if (isProtectedPath(
|
|
298
|
+
if (isProtectedPath(normalizedFilePath)) {
|
|
241
299
|
result.error = `Protected path: "${mod.filePath}" is read-only to the self-improvement loop.`
|
|
242
300
|
result.phase = 'failed'
|
|
243
301
|
this.sessionReport.modifications.push(result)
|
|
@@ -275,7 +333,7 @@ export class CrsiSandbox {
|
|
|
275
333
|
|
|
276
334
|
// Generate diff
|
|
277
335
|
try {
|
|
278
|
-
result.diff = execSync(`git diff -- "${
|
|
336
|
+
result.diff = execSync(`git diff -- "${normalizedFilePath}"`, {
|
|
279
337
|
cwd: this.worktreePath,
|
|
280
338
|
timeout: 10_000,
|
|
281
339
|
encoding: 'utf-8',
|
package/src/core/engine.ts
CHANGED
|
@@ -40,7 +40,12 @@ import { buildRequest, sendInferenceCheck, isInferenceHookEnabled } from './infe
|
|
|
40
40
|
import { getFileInboxTransport } from '../agent/cross-session/file-inbox'
|
|
41
41
|
import { registerWakeupHandler } from '../tools/scheduling/schedule-wakeup'
|
|
42
42
|
import { accumulateGraftSavings } from '../shared/graft-savings'
|
|
43
|
-
import {
|
|
43
|
+
import {
|
|
44
|
+
getMessageBus,
|
|
45
|
+
formatInboundMessage,
|
|
46
|
+
type AgentMessage,
|
|
47
|
+
type AgentMessageBus,
|
|
48
|
+
} from '../agent/message-bus'
|
|
44
49
|
import { createT } from '../i18n-core/t'
|
|
45
50
|
import enUS from '../i18n-core/locales/en-US.json'
|
|
46
51
|
import zhCN from '../i18n-core/locales/zh-CN.json'
|
|
@@ -255,6 +260,28 @@ export class QueryEngine {
|
|
|
255
260
|
}
|
|
256
261
|
}
|
|
257
262
|
|
|
263
|
+
/**
|
|
264
|
+
* Drain unread inbound messages from the in-memory bus (addressed to this
|
|
265
|
+
* session or to "main") and inject them into the conversation as user-role
|
|
266
|
+
* notices. Returns the number of messages injected.
|
|
267
|
+
*
|
|
268
|
+
* The bus is the delivery channel for same-process SendMessage ("main",
|
|
269
|
+
* "bg-*", "sub-agent-*") and the re-post target of pollCrossSessionInbox;
|
|
270
|
+
* this is the missing "drain" step that surfaces those messages to the model.
|
|
271
|
+
*/
|
|
272
|
+
drainInboundMessages(bus: AgentMessageBus = getMessageBus()): number {
|
|
273
|
+
const recipients = [...new Set([this.sessionId, 'main'])]
|
|
274
|
+
let injected = 0
|
|
275
|
+
for (const recipient of recipients) {
|
|
276
|
+
for (const msg of bus.poll(recipient)) {
|
|
277
|
+
this.context.addMessage({ role: 'user', content: formatInboundMessage(msg) })
|
|
278
|
+
injected++
|
|
279
|
+
}
|
|
280
|
+
bus.markAllRead(recipient)
|
|
281
|
+
}
|
|
282
|
+
return injected
|
|
283
|
+
}
|
|
284
|
+
|
|
258
285
|
/** Register a hook engine for pre/post tool-use lifecycle events. */
|
|
259
286
|
setHookEngine(hooks: HookEngine): void {
|
|
260
287
|
this.hookEngine = hooks
|
|
@@ -474,6 +501,8 @@ export class QueryEngine {
|
|
|
474
501
|
|
|
475
502
|
// Poll for cross-session messages before each turn
|
|
476
503
|
await this.pollCrossSessionInbox()
|
|
504
|
+
// Drain in-memory bus messages (same-process SendMessage + cross-session re-posts)
|
|
505
|
+
this.drainInboundMessages()
|
|
477
506
|
|
|
478
507
|
// Fire UserPromptSubmit hooks before processing
|
|
479
508
|
if (this.hookEngine) {
|
|
@@ -1204,7 +1233,6 @@ export class QueryEngine {
|
|
|
1204
1233
|
const ruleId = match?.[1]
|
|
1205
1234
|
if (ruleId) {
|
|
1206
1235
|
tracker.recordApplication(ruleId, result.success, {
|
|
1207
|
-
toolName: name,
|
|
1208
1236
|
error: result.error,
|
|
1209
1237
|
})
|
|
1210
1238
|
}
|