@miphamai/cli 0.68.0 → 0.70.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/commands/loop-scaffold.ts +15 -44
- package/src/config/loader.ts +56 -1
- package/src/core/claude-md-fix.ts +28 -0
- package/src/core/crsi-modify.ts +10 -1
- package/src/core/crsi-producer.ts +13 -0
- package/src/core/engine.ts +31 -0
- package/src/core/eval-harness.ts +94 -2
- package/src/core/fix-code.ts +167 -0
- package/src/core/fix.ts +176 -0
- package/src/core/hooks-config.ts +10 -5
- package/src/core/hooks-executor.ts +84 -5
- package/src/core/instructions.ts +2 -2
- package/src/core/memory/memory-manager.ts +205 -19
- package/src/core/post-flight-checker.ts +102 -0
- package/src/core/reward-fn.ts +3 -1
- package/src/core/session-log.ts +3 -1
- package/src/core/working-memory.ts +140 -0
- package/src/i18n-core/locales/en-US.json +30 -2
- package/src/i18n-core/locales/zh-CN.json +30 -2
- package/src/index.tsx +12 -0
- package/src/shared/package-info.ts +1 -1
- package/src/shared/types.ts +2 -0
- package/src/skills/community-registry.json +16 -0
- package/src/skills/marketplace.ts +233 -0
- package/src/skills/registry.ts +77 -63
- package/src/tools/agent/memory.ts +5 -1
- package/src/tools/exec/task.ts +21 -2
- package/src/ui/commands.ts +354 -34
package/package.json
CHANGED
|
@@ -64,10 +64,6 @@ function writeTemplate(
|
|
|
64
64
|
* ├── .mipham/
|
|
65
65
|
* │ ├── CLAUDE.md
|
|
66
66
|
* │ ├── settings.json
|
|
67
|
-
* │ ├── hooks/
|
|
68
|
-
* │ │ ├── pre-tool-use.sh
|
|
69
|
-
* │ │ ├── post-tool-use.sh
|
|
70
|
-
* │ │ └── stop.sh
|
|
71
67
|
* │ ├── agents/
|
|
72
68
|
* │ │ └── verifier.md
|
|
73
69
|
* │ └── skills/
|
|
@@ -103,13 +99,6 @@ export function scaffoldLoopKit(basePath: string): ScaffoldResult {
|
|
|
103
99
|
// ── .mipham/settings.json ──
|
|
104
100
|
writeTemplate(join(miphamDir, 'settings.json'), TEMPLATES.settingsJson, false, created, skipped)
|
|
105
101
|
|
|
106
|
-
// ── .mipham/hooks/ ──
|
|
107
|
-
const hooksDir = join(miphamDir, 'hooks')
|
|
108
|
-
ensureDir(hooksDir, created, skipped)
|
|
109
|
-
writeTemplate(join(hooksDir, 'pre-tool-use.sh'), TEMPLATES.preToolUse, true, created, skipped)
|
|
110
|
-
writeTemplate(join(hooksDir, 'post-tool-use.sh'), TEMPLATES.postToolUse, true, created, skipped)
|
|
111
|
-
writeTemplate(join(hooksDir, 'stop.sh'), TEMPLATES.stopHook, true, created, skipped)
|
|
112
|
-
|
|
113
102
|
// ── .mipham/agents/ ──
|
|
114
103
|
const agentsDir = join(miphamDir, 'agents')
|
|
115
104
|
ensureDir(agentsDir, created, skipped)
|
|
@@ -180,7 +169,20 @@ const TEMPLATES = {
|
|
|
180
169
|
deny: [],
|
|
181
170
|
},
|
|
182
171
|
hooks: {
|
|
183
|
-
PreToolUse
|
|
172
|
+
// 示例:PreToolUse 拦截 Bash。脚本从 stdin 读 JSON(tool_name/tool_input),
|
|
173
|
+
// 输出 hookSpecificOutput JSON 决定 allow/deny;exit 2 = 拦截(stderr 作理由)。
|
|
174
|
+
PreToolUse: [
|
|
175
|
+
{
|
|
176
|
+
matcher: 'Bash',
|
|
177
|
+
hooks: [
|
|
178
|
+
{
|
|
179
|
+
type: 'command',
|
|
180
|
+
command: 'your-hook-script.sh',
|
|
181
|
+
timeout: 60,
|
|
182
|
+
},
|
|
183
|
+
],
|
|
184
|
+
},
|
|
185
|
+
],
|
|
184
186
|
PostToolUse: [],
|
|
185
187
|
Stop: [],
|
|
186
188
|
SessionStart: [],
|
|
@@ -192,37 +194,6 @@ const TEMPLATES = {
|
|
|
192
194
|
2,
|
|
193
195
|
) + '\n',
|
|
194
196
|
|
|
195
|
-
preToolUse: `#!/bin/bash
|
|
196
|
-
# PreToolUse hook — runs before each tool execution.
|
|
197
|
-
# Tool name passed as \$1, input JSON as \$2.
|
|
198
|
-
# Exit non-zero to block the tool.
|
|
199
|
-
# Write JSON to stdout to modify the tool input.
|
|
200
|
-
|
|
201
|
-
TOOL_NAME="\$1"
|
|
202
|
-
TOOL_INPUT="\$2"
|
|
203
|
-
|
|
204
|
-
echo "{\\"decision\\": \\"allow\\"}" >&2
|
|
205
|
-
exit 0
|
|
206
|
-
`,
|
|
207
|
-
|
|
208
|
-
postToolUse: `#!/bin/bash
|
|
209
|
-
# PostToolUse hook — runs after each tool execution.
|
|
210
|
-
# Tool name passed as \$1, result JSON as \$2.
|
|
211
|
-
|
|
212
|
-
TOOL_NAME="\$1"
|
|
213
|
-
TOOL_RESULT="\$2"
|
|
214
|
-
|
|
215
|
-
exit 0
|
|
216
|
-
`,
|
|
217
|
-
|
|
218
|
-
stopHook: `#!/bin/bash
|
|
219
|
-
# Stop hook — runs when the AI session ends.
|
|
220
|
-
# Use for cleanup, notifications, or saving state.
|
|
221
|
-
|
|
222
|
-
echo "Session ended at \$(date)" >&2
|
|
223
|
-
exit 0
|
|
224
|
-
`,
|
|
225
|
-
|
|
226
197
|
verifierAgent: `# Verifier Agent
|
|
227
198
|
|
|
228
199
|
> Pre-commit audit specialist. Dispatched BEFORE committing to verify changes.
|
|
@@ -290,7 +261,7 @@ Verify staged changes against coding standards, security rules, and best practic
|
|
|
290
261
|
|
|
291
262
|
## Structure
|
|
292
263
|
|
|
293
|
-
- \`.mipham/\` — Mipham Code configuration (CLAUDE.md, settings,
|
|
264
|
+
- \`.mipham/\` — Mipham Code configuration (CLAUDE.md, settings.json, agents, skills)
|
|
294
265
|
- \`.mcp.json\` — MCP server configuration
|
|
295
266
|
- \`MEMORY.md\` — AI persistent memory
|
|
296
267
|
- \`run.sh\` — Project launcher
|
package/src/config/loader.ts
CHANGED
|
@@ -28,6 +28,7 @@ import {
|
|
|
28
28
|
DEFAULT_CROSS_SESSION_CONFIG,
|
|
29
29
|
} from './defaults'
|
|
30
30
|
import { getCredentialKey, encryptApiKey, decryptApiKey, ENC_PREFIX } from './credential-crypto'
|
|
31
|
+
import type { SettingsHooks } from '../core/hooks-config'
|
|
31
32
|
|
|
32
33
|
const MIPHAM_HOME = join(homedir(), '.mipham')
|
|
33
34
|
const BACKUP_PREFIX = 'config.backup-'
|
|
@@ -139,7 +140,7 @@ function backupConfig(configPath: string): void {
|
|
|
139
140
|
* Try to restore config from the most recent backup.
|
|
140
141
|
* Returns true if restored successfully.
|
|
141
142
|
*/
|
|
142
|
-
function tryRestoreFromBackup(configPath: string): boolean {
|
|
143
|
+
export function tryRestoreFromBackup(configPath: string): boolean {
|
|
143
144
|
try {
|
|
144
145
|
if (!existsSync(MIPHAM_HOME)) return false
|
|
145
146
|
const files = readdirSync(MIPHAM_HOME)
|
|
@@ -216,6 +217,60 @@ function loadMcpJson(cwd: string): McpServerConfig[] {
|
|
|
216
217
|
return servers
|
|
217
218
|
}
|
|
218
219
|
|
|
220
|
+
/**
|
|
221
|
+
* Parsed `settings.json` (Claude Code convention): hooks + permissions.
|
|
222
|
+
* Hooks are additive across levels; permissions allow/deny are deduped unions.
|
|
223
|
+
*/
|
|
224
|
+
export interface SettingsJson {
|
|
225
|
+
hooks: SettingsHooks
|
|
226
|
+
permissions: { allow: string[]; deny: string[] }
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Load `settings.json` — project-level `.mipham/settings.json` then user-level
|
|
231
|
+
* `~/.mipham/settings.json`. Mirrors the Claude Code convention (hooks additive,
|
|
232
|
+
* permissions merged), so users can migrate their Claude settings unchanged.
|
|
233
|
+
*/
|
|
234
|
+
export function loadSettingsJson(cwd: string = process.cwd()): SettingsJson {
|
|
235
|
+
const hooks: SettingsHooks = {}
|
|
236
|
+
const permissions = { allow: [] as string[], deny: [] as string[] }
|
|
237
|
+
|
|
238
|
+
const searchPaths = [join(cwd, '.mipham', 'settings.json'), join(MIPHAM_HOME, 'settings.json')]
|
|
239
|
+
|
|
240
|
+
for (const path of searchPaths) {
|
|
241
|
+
try {
|
|
242
|
+
if (!existsSync(path)) continue
|
|
243
|
+
const raw = readFileSync(path, 'utf-8')
|
|
244
|
+
const parsed = JSON.parse(raw) as {
|
|
245
|
+
hooks?: Record<string, unknown>
|
|
246
|
+
permissions?: { allow?: unknown; deny?: unknown }
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
if (parsed.hooks && typeof parsed.hooks === 'object') {
|
|
250
|
+
for (const [eventName, entries] of Object.entries(parsed.hooks)) {
|
|
251
|
+
if (!Array.isArray(entries)) continue
|
|
252
|
+
const bucket = (hooks as Record<string, unknown[]>)[eventName]
|
|
253
|
+
;(hooks as Record<string, unknown[]>)[eventName] = [...(bucket ?? []), ...entries]
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
if (parsed.permissions) {
|
|
258
|
+
for (const key of ['allow', 'deny'] as const) {
|
|
259
|
+
const list = parsed.permissions[key]
|
|
260
|
+
if (!Array.isArray(list)) continue
|
|
261
|
+
for (const p of list) {
|
|
262
|
+
if (typeof p === 'string' && !permissions[key].includes(p)) permissions[key].push(p)
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
} catch {
|
|
267
|
+
// Silently skip malformed or missing settings.json files
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
return { hooks, permissions }
|
|
272
|
+
}
|
|
273
|
+
|
|
219
274
|
export function loadConfig(cwd: string = process.cwd()): MiphamConfig {
|
|
220
275
|
const configPath = join(cwd, '.mipham', 'config.yml')
|
|
221
276
|
const userConfigPath = join(MIPHAM_HOME, 'config.yml')
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CLAUDE.md fix — write the derivable-section headings that `findDerivableSections`
|
|
3
|
+
* flags into the document's `prompt-exclude` frontmatter, so `stripSections` stops
|
|
4
|
+
* injecting them into the system prompt every session.
|
|
5
|
+
*/
|
|
6
|
+
import { stringify as stringifyYaml } from 'yaml'
|
|
7
|
+
import { parseFrontmatter, parsePromptExclude } from './instructions'
|
|
8
|
+
|
|
9
|
+
export interface ApplyPromptExcludeResult {
|
|
10
|
+
content: string
|
|
11
|
+
added: string[]
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Merge `headings` into the document's `prompt-exclude` frontmatter (creating the
|
|
16
|
+
* frontmatter block if absent). Preserves existing exclusions and other frontmatter
|
|
17
|
+
* fields; reports only the headings that were actually new.
|
|
18
|
+
*/
|
|
19
|
+
export function applyPromptExclude(content: string, headings: string[]): ApplyPromptExcludeResult {
|
|
20
|
+
const { data, content: body } = parseFrontmatter(content)
|
|
21
|
+
const existing = parsePromptExclude(data['prompt-exclude'])
|
|
22
|
+
const added = [...new Set(headings)].filter((h) => !existing.includes(h))
|
|
23
|
+
if (added.length === 0) return { content, added: [] }
|
|
24
|
+
|
|
25
|
+
data['prompt-exclude'] = [...existing, ...added]
|
|
26
|
+
const frontmatter = stringifyYaml(data)
|
|
27
|
+
return { content: `---\n${frontmatter}---\n${body}`, added }
|
|
28
|
+
}
|
package/src/core/crsi-modify.ts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
16
|
import { CrsiSandbox, validateBlastRadius } from './crsi-sandbox'
|
|
17
17
|
import type { CrsiModificationResult } from './crsi-sandbox'
|
|
18
|
-
import { appendEvalScore, getLastEvalScore } from './eval-harness'
|
|
18
|
+
import { appendEvalScore, getLastEvalScore, regressedAnchors } from './eval-harness'
|
|
19
19
|
import { mechanismSentinel, type RewardFn } from './reward-fn'
|
|
20
20
|
|
|
21
21
|
export interface CrsiProposal {
|
|
@@ -102,6 +102,15 @@ export async function runCrsiModification(
|
|
|
102
102
|
// 默认机制哨兵;可插拔——调用方传 opts.rewardFn 换用其他奖励源(如任务表现)。
|
|
103
103
|
const rewardFn = opts?.rewardFn ?? mechanismSentinel()
|
|
104
104
|
const report = await rewardFn.evaluate()
|
|
105
|
+
// 细粒度防回退:anchor 契约(安全/机制不变量)任一翻转 PASS→FAIL 即拒,
|
|
106
|
+
// 即使总分因新增契约而上升也被拦(比「总分不退化」更严格)。
|
|
107
|
+
const anchors = regressedAnchors(report.results ?? [])
|
|
108
|
+
if (anchors.length > 0) {
|
|
109
|
+
sandbox.rollback()
|
|
110
|
+
applied.phase = 'failed'
|
|
111
|
+
applied.error = `Anchor regression: ${anchors.join(', ')}`
|
|
112
|
+
return applied
|
|
113
|
+
}
|
|
105
114
|
const last = getLastEvalScore(rewardFn.name)
|
|
106
115
|
if (last !== null && report.score < last) {
|
|
107
116
|
sandbox.rollback()
|
|
@@ -20,6 +20,9 @@ import { homedir } from 'node:os'
|
|
|
20
20
|
/** 教训文件(相对仓库根)。预建,沙箱只能改已存在文件。 */
|
|
21
21
|
export const LESSONS_FILE = 'apps/cli/crsi-lessons.md'
|
|
22
22
|
|
|
23
|
+
/** 组件归因(修复决策,非因果断言):失败最可能被哪个记忆组件的局部干预修复。 */
|
|
24
|
+
export type MemoryComponent = 'experiential' | 'working' | 'invocation' | 'checker'
|
|
25
|
+
|
|
23
26
|
/** 归一化的教训信号(insight 与 meta-rule 的公共面)。 */
|
|
24
27
|
export interface CrsiSignal {
|
|
25
28
|
category: string
|
|
@@ -27,6 +30,12 @@ export interface CrsiSignal {
|
|
|
27
30
|
severity?: string
|
|
28
31
|
suggestion: string
|
|
29
32
|
evidence: string[]
|
|
33
|
+
/**
|
|
34
|
+
* 组件归因:该失败最可能被哪个记忆组件的局部干预修复。
|
|
35
|
+
* 缺省 experiential(现状 = 教训/技能/受管理规则,全属 E)。
|
|
36
|
+
* working/invocation/checker 待 ②③ 落地后由对应信号源产出。
|
|
37
|
+
*/
|
|
38
|
+
component?: MemoryComponent
|
|
30
39
|
}
|
|
31
40
|
|
|
32
41
|
const SEVERITY_RANK: Record<string, number> = { critical: 0, warning: 1, info: 2 }
|
|
@@ -76,6 +85,7 @@ export function buildLessonContent(
|
|
|
76
85
|
`## ${signal.category}: ${signal.title}`,
|
|
77
86
|
'',
|
|
78
87
|
`- 建议: ${signal.suggestion}`,
|
|
88
|
+
`- 组件: ${signal.component ?? 'experiential'}`,
|
|
79
89
|
]
|
|
80
90
|
if (signal.severity) lines.push(`- 严重度: ${signal.severity}`)
|
|
81
91
|
lines.push(`- 生成时间: ${timestamp}`, `- 来源: ${source}`, '', '### 证据')
|
|
@@ -188,6 +198,9 @@ export function managedRuleId(signal: CrsiSignal): string {
|
|
|
188
198
|
* 只支持 timeout / tool-params 两类确定性 category,其余返回 null。
|
|
189
199
|
*/
|
|
190
200
|
export function renderManagedRuleSource(signal: CrsiSignal): string | null {
|
|
201
|
+
// managed-rule 是「行为固化」,只处理 experiential 组件(timeout/tool-params 本质是 E)。
|
|
202
|
+
// working/invocation/checker 的修复走各自组件路径,不进这里。
|
|
203
|
+
if ((signal.component ?? 'experiential') !== 'experiential') return null
|
|
191
204
|
const id = managedRuleId(signal)
|
|
192
205
|
const warning = signal.suggestion || `CRSI 自动固化: ${signal.title}`
|
|
193
206
|
|
package/src/core/engine.ts
CHANGED
|
@@ -29,6 +29,8 @@ import { CrsiProvenanceBridge } from '../agent/crsi-provenance-bridge.js'
|
|
|
29
29
|
import { McpClient } from '../mcp/client.js'
|
|
30
30
|
import { ErrorSignatureDB } from './error-signature-db.js'
|
|
31
31
|
import { PreFlightChecker } from './preflight-checker.js'
|
|
32
|
+
import { PostFlightChecker, createDefaultPostFlightChecker } from './post-flight-checker'
|
|
33
|
+
import { recordToolEvidence } from './working-memory'
|
|
32
34
|
import { AutoCorrector } from './auto-corrector.js'
|
|
33
35
|
import { MetaRuleEngine } from './meta-rule-engine.js'
|
|
34
36
|
import { DreamEngine } from './dream-engine.js'
|
|
@@ -83,6 +85,7 @@ export class QueryEngine {
|
|
|
83
85
|
private _autoMemory?: AutoMemoryEngine
|
|
84
86
|
private _errorSignatureDB?: ErrorSignatureDB
|
|
85
87
|
private _preflightChecker?: PreFlightChecker
|
|
88
|
+
private _postFlightChecker?: PostFlightChecker
|
|
86
89
|
private _autoCorrector?: AutoCorrector
|
|
87
90
|
private _metaRuleEngine?: MetaRuleEngine
|
|
88
91
|
private _dreamEngine?: DreamEngine
|
|
@@ -1243,6 +1246,23 @@ export class QueryEngine {
|
|
|
1243
1246
|
// Accumulate graft token savings from "[graft] tokens saved ≈ N" footers
|
|
1244
1247
|
accumulateGraftSavings(result.content)
|
|
1245
1248
|
|
|
1249
|
+
// ── PostFlightChecker — 事后验证「观察是否支撑变更」(Recuris C 组件)──
|
|
1250
|
+
// 默认 no-checker 静默、rejected 只记录不阻塞(第一阶段只产证据,不强制拦截)。
|
|
1251
|
+
const decision = this.getPostFlightChecker().check(name, {
|
|
1252
|
+
params: effectiveParams,
|
|
1253
|
+
result,
|
|
1254
|
+
})
|
|
1255
|
+
// 证据账本:supported/rejected 回灌工作记忆(供 Task 完成门读取)
|
|
1256
|
+
recordToolEvidence(name, decision)
|
|
1257
|
+
if (decision.verdict !== 'no-checker') {
|
|
1258
|
+
this.context.getLog()?.append({
|
|
1259
|
+
type: 'checker/decision',
|
|
1260
|
+
at: Date.now(),
|
|
1261
|
+
toolName: name,
|
|
1262
|
+
decision,
|
|
1263
|
+
})
|
|
1264
|
+
}
|
|
1265
|
+
|
|
1246
1266
|
return result
|
|
1247
1267
|
} catch (err) {
|
|
1248
1268
|
// P1-3: Trigger PostToolUseFailure hook on tool execution errors
|
|
@@ -1488,6 +1508,17 @@ export class QueryEngine {
|
|
|
1488
1508
|
return this._preflightChecker
|
|
1489
1509
|
}
|
|
1490
1510
|
|
|
1511
|
+
/**
|
|
1512
|
+
* Recuris C 组件:Lazily-initialized PostFlightChecker singleton。
|
|
1513
|
+
* 事后验证「工具/env 观察是否支撑变更」,与 PreFlightChecker(事前拦截)互补。
|
|
1514
|
+
*/
|
|
1515
|
+
getPostFlightChecker(): PostFlightChecker {
|
|
1516
|
+
if (!this._postFlightChecker) {
|
|
1517
|
+
this._postFlightChecker = createDefaultPostFlightChecker()
|
|
1518
|
+
}
|
|
1519
|
+
return this._postFlightChecker
|
|
1520
|
+
}
|
|
1521
|
+
|
|
1491
1522
|
/**
|
|
1492
1523
|
* SIS Phase 2: Lazily-initialized AutoCorrector singleton.
|
|
1493
1524
|
* Analyzes failed tool calls and suggests corrections based on
|
package/src/core/eval-harness.ts
CHANGED
|
@@ -18,19 +18,31 @@ import { ExperienceRuleEngine } from './rule-engine'
|
|
|
18
18
|
import { ConstitutionLoader, DEFAULT_CONSTITUTION } from './constitution-loader'
|
|
19
19
|
import { ErrorSignatureDB } from './error-signature-db'
|
|
20
20
|
import { PreFlightChecker } from './preflight-checker'
|
|
21
|
+
import { createDefaultPostFlightChecker } from './post-flight-checker'
|
|
22
|
+
import { WorkingMemory } from './working-memory'
|
|
21
23
|
import { RedTeam } from './red-team'
|
|
22
24
|
import { isProtectedPath, validateBlastRadius, PROTECTED_CRITICAL_FILES } from './crsi-sandbox'
|
|
23
|
-
import {
|
|
25
|
+
import {
|
|
26
|
+
produceRuleProposal,
|
|
27
|
+
MANAGED_RULES_FILE,
|
|
28
|
+
buildLessonContent,
|
|
29
|
+
renderManagedRuleSource,
|
|
30
|
+
} from './crsi-producer'
|
|
24
31
|
import type { CrsiSignal } from './crsi-producer'
|
|
25
32
|
import { loadBehaviorTasks, judgeBehaviorTask } from './behavior-tasks'
|
|
26
33
|
|
|
27
34
|
// ── Types ──
|
|
28
35
|
|
|
36
|
+
/** 契约角色:anchor = 安全/机制不变量(门强制不许回退);target = 缺口/覆盖(应被补)。 */
|
|
37
|
+
export type ContractRole = 'anchor' | 'target' | 'neutral'
|
|
38
|
+
|
|
29
39
|
export interface EvalResult {
|
|
30
40
|
id: string
|
|
31
41
|
description: string
|
|
32
42
|
passed: boolean
|
|
33
43
|
detail?: string
|
|
44
|
+
/** 契约角色。缺省 neutral。 */
|
|
45
|
+
role?: ContractRole
|
|
34
46
|
}
|
|
35
47
|
|
|
36
48
|
export interface EvalReport {
|
|
@@ -42,6 +54,29 @@ export interface EvalReport {
|
|
|
42
54
|
failures: string[]
|
|
43
55
|
}
|
|
44
56
|
|
|
57
|
+
/** anchor 契约 id 集合:安全/机制不变量,绝不许回退。门(crsi-modify)强制此集合零回退。 */
|
|
58
|
+
export const ANCHOR_CONTRACT_IDS: ReadonlySet<string> = new Set([
|
|
59
|
+
'rule-timeout',
|
|
60
|
+
'rule-git-force',
|
|
61
|
+
'rule-disabled-skip',
|
|
62
|
+
'constitution-8-principles',
|
|
63
|
+
'constitution-facets',
|
|
64
|
+
'constitution-preamble',
|
|
65
|
+
'sandbox-protected-constitution',
|
|
66
|
+
'sandbox-protected-tests',
|
|
67
|
+
'sandbox-protected-machinery',
|
|
68
|
+
'protection-completeness',
|
|
69
|
+
'blast-radius-gate',
|
|
70
|
+
'red-team-zero-gaps',
|
|
71
|
+
'producer-rule-shape',
|
|
72
|
+
'producer-rule-idempotent',
|
|
73
|
+
])
|
|
74
|
+
|
|
75
|
+
/** 细粒度防回退:返回 role==='anchor' 且已 FAIL 的契约 id。空 = 无 anchor 回退。 */
|
|
76
|
+
export function regressedAnchors(results: EvalResult[]): string[] {
|
|
77
|
+
return results.filter((r) => r.role === 'anchor' && !r.passed).map((r) => r.id)
|
|
78
|
+
}
|
|
79
|
+
|
|
45
80
|
// ── Rewards log (path A Phase 1: 奖励信号持久化) ──
|
|
46
81
|
|
|
47
82
|
const SCORES_FILE = join(homedir(), '.mipham', 'crsi', 'eval-scores.jsonl')
|
|
@@ -228,6 +263,48 @@ export function runEval(): EvalReport {
|
|
|
228
263
|
ruleProposal !== null && produceRuleProposal(frozenSignal, ruleProposal.newContent) === null,
|
|
229
264
|
})
|
|
230
265
|
|
|
266
|
+
// ── 组件归因(ground truth:缺省 experiential、显式组件透传、非 experiential 不进 managed-rule) ──
|
|
267
|
+
results.push({
|
|
268
|
+
id: 'producer-component-tag',
|
|
269
|
+
description: '组件归因:缺省 experiential、显式组件透传、非 experiential 不进 managed-rule',
|
|
270
|
+
passed:
|
|
271
|
+
buildLessonContent(frozenSignal, 't', 'src').includes('- 组件: experiential') &&
|
|
272
|
+
buildLessonContent({ ...frozenSignal, component: 'checker' }, 't', 'src').includes(
|
|
273
|
+
'- 组件: checker',
|
|
274
|
+
) &&
|
|
275
|
+
renderManagedRuleSource({ ...frozenSignal, component: 'working' }) === null &&
|
|
276
|
+
renderManagedRuleSource(frozenSignal) !== null,
|
|
277
|
+
})
|
|
278
|
+
|
|
279
|
+
// ── 事后检查器(ground truth:exit 0 判 supported、exit 非 0 判 rejected) ──
|
|
280
|
+
const postFlight = createDefaultPostFlightChecker()
|
|
281
|
+
results.push({
|
|
282
|
+
id: 'postflight-bash-exit',
|
|
283
|
+
description: '事后检查器:bash exit 0 判 supported、exit 非 0 判 rejected',
|
|
284
|
+
passed:
|
|
285
|
+
postFlight.check('Bash', { params: {}, result: { success: true, content: '' } }).verdict ===
|
|
286
|
+
'supported' &&
|
|
287
|
+
postFlight.check('Bash', { params: {}, result: { success: false, content: '', error: 'x' } })
|
|
288
|
+
.verdict === 'rejected',
|
|
289
|
+
})
|
|
290
|
+
|
|
291
|
+
// ── 工作记忆证据接地(ground truth:done 只能由 supported 推进,模型自称不算) ──
|
|
292
|
+
const wm = new WorkingMemory()
|
|
293
|
+
wm.setGoal('install-deps', 'install dependencies')
|
|
294
|
+
wm.observe('install-deps', { verdict: 'no-checker' })
|
|
295
|
+
const pendingAfterNoChecker = wm.getGoal('install-deps')!.status === 'pending'
|
|
296
|
+
wm.observe('install-deps', { verdict: 'supported', checkerId: 'bash-exit' })
|
|
297
|
+
const doneAfterSupported = wm.getGoal('install-deps')!.status === 'done'
|
|
298
|
+
wm.setGoal('edit-file', 'edit the file')
|
|
299
|
+
wm.observe('edit-file', { verdict: 'rejected', checkerId: 'edit-applied', reason: 'x' })
|
|
300
|
+
const blockedAfterRejected = wm.getGoal('edit-file')!.status === 'blocked'
|
|
301
|
+
results.push({
|
|
302
|
+
id: 'working-memory-evidence-gated',
|
|
303
|
+
description:
|
|
304
|
+
'工作记忆:done 只能由 checker supported 推进,rejected 置 blocked,模型自称(no-checker)不算',
|
|
305
|
+
passed: pendingAfterNoChecker && doneAfterSupported && blockedAfterRejected,
|
|
306
|
+
})
|
|
307
|
+
|
|
231
308
|
// ── 行为缺口(ground truth:当前无规则覆盖的确定性拦截,如实判 FAIL) ──
|
|
232
309
|
// producer 固化 tool-params 规则后,这些缺口翻转 PASS → 分数上升 =「证明更好」。
|
|
233
310
|
const behaviorGaps: Array<{ id: string; command: string }> = [
|
|
@@ -246,15 +323,30 @@ export function runEval(): EvalReport {
|
|
|
246
323
|
id: gap.id,
|
|
247
324
|
description: `行为缺口未覆盖: ${gap.command}`,
|
|
248
325
|
passed: r.warnings.length > 0,
|
|
326
|
+
role: 'target',
|
|
249
327
|
})
|
|
250
328
|
}
|
|
251
329
|
|
|
252
330
|
// ── 行为任务集(ground truth:约束行为效果,确定性无 LLM) ──
|
|
253
331
|
const behaviorTasks = loadBehaviorTasks()
|
|
254
332
|
for (const task of behaviorTasks) {
|
|
255
|
-
results.push(judgeBehaviorTask(task, ruleEngine))
|
|
333
|
+
results.push({ ...judgeBehaviorTask(task, ruleEngine), role: 'target' })
|
|
256
334
|
}
|
|
257
335
|
|
|
336
|
+
// 角色标注:anchor 走集中清单(门保护面单一真源),target 已在上方循环内联。
|
|
337
|
+
for (const r of results) {
|
|
338
|
+
if (ANCHOR_CONTRACT_IDS.has(r.id)) r.role = 'anchor'
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
// anchor 自检(ground truth:所有 anchor 契约必须全绿,否则门拒)。
|
|
342
|
+
const anchorFailures = results.filter((r) => r.role === 'anchor' && !r.passed).map((r) => r.id)
|
|
343
|
+
results.push({
|
|
344
|
+
id: 'anchor-gate',
|
|
345
|
+
description: '所有 anchor 契约必须全绿(细粒度防回退闸)',
|
|
346
|
+
passed: anchorFailures.length === 0,
|
|
347
|
+
...(anchorFailures.length > 0 ? { detail: `回退的 anchor: ${anchorFailures.join(', ')}` } : {}),
|
|
348
|
+
})
|
|
349
|
+
|
|
258
350
|
const passed = results.filter((r) => r.passed).length
|
|
259
351
|
return {
|
|
260
352
|
total: results.length,
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `/fix test` — LLM-driven repair of failing tests in the real repo.
|
|
3
|
+
*
|
|
4
|
+
* Reuses the `/crsi bench` closed loop (LLM generates → a frozen test judges the
|
|
5
|
+
* result) but applied to the caller's own codebase: the failing test is the frozen
|
|
6
|
+
* ground truth, and the LLM may only fix the *source*, never the test.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Extract failing test-file paths from a `vitest run` output stream. Matches the
|
|
11
|
+
* ` FAIL test/xxx.test.ts > describe > it` lines (vitest 4 default reporter).
|
|
12
|
+
*/
|
|
13
|
+
export function parseVitestFailures(output: string): string[] {
|
|
14
|
+
const files = new Set<string>()
|
|
15
|
+
const re = /^\s*FAIL\s+(\S+)/gm
|
|
16
|
+
let m
|
|
17
|
+
while ((m = re.exec(output))) {
|
|
18
|
+
files.add(m[1]!)
|
|
19
|
+
}
|
|
20
|
+
return [...files]
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Extract relative import specifiers (`./foo`, `../bar`) from a test file — these
|
|
25
|
+
* are the candidate source files the failing test depends on. Package imports
|
|
26
|
+
* (`vitest`, `lodash`) and node builtins (`node:fs`) are ignored.
|
|
27
|
+
*/
|
|
28
|
+
export function collectLocalImports(testContent: string): string[] {
|
|
29
|
+
const imports: string[] = []
|
|
30
|
+
const re = /from\s+['"](\.[^'"]*)['"]/g
|
|
31
|
+
let m
|
|
32
|
+
while ((m = re.exec(testContent))) {
|
|
33
|
+
imports.push(m[1]!)
|
|
34
|
+
}
|
|
35
|
+
return imports
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Build the LLM prompt that asks for a source-only fix. The test is declared
|
|
40
|
+
* frozen ground truth; the model must return the corrected source file contents.
|
|
41
|
+
*/
|
|
42
|
+
export function buildFixPrompt(opts: {
|
|
43
|
+
testPath: string
|
|
44
|
+
testContent: string
|
|
45
|
+
sourcePath: string
|
|
46
|
+
sourceContent: string
|
|
47
|
+
failure: string
|
|
48
|
+
}): string {
|
|
49
|
+
const { testPath, testContent, sourcePath, sourceContent, failure } = opts
|
|
50
|
+
return [
|
|
51
|
+
'You are fixing a failing test in a real codebase.',
|
|
52
|
+
'',
|
|
53
|
+
'RULES:',
|
|
54
|
+
'- The test file is FROZEN ground truth. Do NOT modify the test.',
|
|
55
|
+
'- Only fix the SOURCE file so the test passes.',
|
|
56
|
+
'- Output ONLY the corrected source file contents — no markdown fences, no explanation.',
|
|
57
|
+
'',
|
|
58
|
+
`Failing test: ${testPath}`,
|
|
59
|
+
'```',
|
|
60
|
+
testContent,
|
|
61
|
+
'```',
|
|
62
|
+
'',
|
|
63
|
+
`Source to fix: ${sourcePath}`,
|
|
64
|
+
'```',
|
|
65
|
+
sourceContent,
|
|
66
|
+
'```',
|
|
67
|
+
'',
|
|
68
|
+
'Failure:',
|
|
69
|
+
failure,
|
|
70
|
+
].join('\n')
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface FixCodeDeps {
|
|
74
|
+
runVitest: (testFile: string) => { exitCode: number; output: string }
|
|
75
|
+
readFile: (path: string) => string | null
|
|
76
|
+
writeFile: (path: string, content: string) => void
|
|
77
|
+
generateFix: (prompt: string) => Promise<string>
|
|
78
|
+
resolveSourceFile: (testFile: string, specifier: string) => string
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export interface FixTargetResult {
|
|
82
|
+
testFile: string
|
|
83
|
+
sourceFile: string | null
|
|
84
|
+
fixed: boolean
|
|
85
|
+
attempts: number
|
|
86
|
+
detail?: string
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Repair one failing test: locate the first local source the test imports, ask
|
|
91
|
+
* the LLM for a source-only fix, verify it against the frozen test, and apply it
|
|
92
|
+
* (or restore the original in dry-run). Retries up to `maxRetries`, feeding each
|
|
93
|
+
* failure back into the next prompt.
|
|
94
|
+
*/
|
|
95
|
+
export async function fixCodeTarget(
|
|
96
|
+
deps: FixCodeDeps,
|
|
97
|
+
testFile: string,
|
|
98
|
+
opts?: { apply?: boolean; maxRetries?: number },
|
|
99
|
+
): Promise<FixTargetResult> {
|
|
100
|
+
const apply = opts?.apply ?? false
|
|
101
|
+
const maxRetries = opts?.maxRetries ?? 3
|
|
102
|
+
|
|
103
|
+
const initial = deps.runVitest(testFile)
|
|
104
|
+
if (initial.exitCode === 0) {
|
|
105
|
+
return { testFile, sourceFile: null, fixed: true, attempts: 0, detail: 'test already passes' }
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const testContent = deps.readFile(testFile)
|
|
109
|
+
if (testContent === null) {
|
|
110
|
+
return {
|
|
111
|
+
testFile,
|
|
112
|
+
sourceFile: null,
|
|
113
|
+
fixed: false,
|
|
114
|
+
attempts: 0,
|
|
115
|
+
detail: 'cannot read test file',
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const specifiers = collectLocalImports(testContent)
|
|
120
|
+
if (specifiers.length === 0) {
|
|
121
|
+
return {
|
|
122
|
+
testFile,
|
|
123
|
+
sourceFile: null,
|
|
124
|
+
fixed: false,
|
|
125
|
+
attempts: 0,
|
|
126
|
+
detail: 'no local imports to locate a source file',
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const sourceFile = deps.resolveSourceFile(testFile, specifiers[0]!)
|
|
131
|
+
const sourceContent = deps.readFile(sourceFile)
|
|
132
|
+
if (sourceContent === null) {
|
|
133
|
+
return { testFile, sourceFile, fixed: false, attempts: 0, detail: 'cannot read source file' }
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
let failure = initial.output
|
|
137
|
+
for (let attempt = 1; attempt <= maxRetries; attempt++) {
|
|
138
|
+
const prompt = buildFixPrompt({
|
|
139
|
+
testPath: testFile,
|
|
140
|
+
testContent,
|
|
141
|
+
sourcePath: sourceFile,
|
|
142
|
+
sourceContent,
|
|
143
|
+
failure,
|
|
144
|
+
})
|
|
145
|
+
const fixedContent = await deps.generateFix(prompt)
|
|
146
|
+
if (!fixedContent) continue
|
|
147
|
+
|
|
148
|
+
deps.writeFile(sourceFile, fixedContent)
|
|
149
|
+
const result = deps.runVitest(testFile)
|
|
150
|
+
|
|
151
|
+
if (result.exitCode === 0) {
|
|
152
|
+
if (!apply) deps.writeFile(sourceFile, sourceContent)
|
|
153
|
+
return { testFile, sourceFile, fixed: true, attempts: attempt }
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
deps.writeFile(sourceFile, sourceContent)
|
|
157
|
+
failure = result.output
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
testFile,
|
|
162
|
+
sourceFile,
|
|
163
|
+
fixed: false,
|
|
164
|
+
attempts: maxRetries,
|
|
165
|
+
detail: 'fix never passed the frozen test',
|
|
166
|
+
}
|
|
167
|
+
}
|