@miphamai/cli 0.37.0 → 0.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/mipham.ts +25 -0
- package/package.json +1 -1
- package/src/agent/cross-session/discovery.ts +45 -1
- package/src/agent/cross-session/file-inbox.ts +15 -0
- package/src/agent/message-router.ts +22 -2
- package/src/agent/sub-agent.ts +4 -3
- package/src/config/defaults.ts +19 -1
- package/src/config/keys-manager.ts +11 -1
- package/src/config/loader.ts +20 -10
- package/src/core/context.ts +86 -56
- package/src/core/credential-masker/output-scrub.ts +5 -0
- package/src/core/crsi-sandbox.ts +19 -7
- package/src/core/engine.ts +69 -35
- package/src/core/instructions.ts +4 -42
- package/src/core/permission.ts +7 -1
- package/src/core/red-team.ts +20 -5
- package/src/core/self-critique.ts +3 -1
- package/src/core/session-log.ts +225 -0
- package/src/core/session-store.ts +119 -34
- package/src/core/workspace-trust.ts +25 -2
- package/src/daemon/auth.ts +9 -6
- package/src/daemon/cors.ts +24 -16
- package/src/daemon/rate-limiter.ts +4 -0
- package/src/daemon/server.ts +24 -3
- package/src/i18n-core/locales/en-US.json +4 -0
- package/src/i18n-core/locales/zh-CN.json +4 -0
- package/src/index.tsx +62 -30
- package/src/mcp/client.ts +56 -17
- package/src/mcp/http-transport.ts +2 -0
- package/src/mcp/oauth.ts +1 -1
- package/src/providers/llm-replay.ts +36 -0
- package/src/providers/llm.ts +16 -0
- package/src/providers/registry.ts +2 -1
- package/src/security/gate.ts +10 -0
- package/src/security/url.ts +37 -14
- package/src/shared/constants.ts +4 -486
- package/src/shared/types.ts +6 -437
- package/src/skills/fork-executor.ts +3 -1
- package/src/skills/loader.ts +2 -1
- package/src/skills/seam.ts +18 -0
- package/src/tools/agent/agent.ts +1 -0
- package/src/tools/agent/skill.ts +1 -0
- package/src/tools/exec/bash.ts +141 -129
- package/src/tools/exec/git.ts +17 -0
- package/src/tools/file/read.ts +73 -66
- package/src/tools/file/write.ts +1 -1
- package/src/tools/index.ts +54 -132
- package/src/tools/network/web-fetch.ts +41 -26
- package/src/tools/seam.ts +25 -0
- package/src/tools/validation.ts +91 -0
- package/src/ui/app.tsx +36 -0
- package/src/ui/commands.ts +27 -6
- package/src/ui/input.tsx +21 -0
- package/src/vajra/compose/assemble.ts +19 -0
- package/src/vajra/compose/bundle.ts +39 -0
- package/src/vajra/compose/dump.ts +5 -0
- package/src/vajra/compose/index.ts +4 -0
- package/src/vajra/compose/mount.ts +30 -0
- package/src/vajra/context.ts +183 -0
- package/src/vajra/events.ts +13 -0
- package/src/vajra/index.ts +4 -0
- package/src/vajra/leaf/plan-runner.ts +94 -0
- package/src/vajra/service.ts +15 -0
- package/src/workflow/primitives/agent.ts +3 -1
- package/src/workflow/runtime.ts +23 -15
- package/src/workflow/sandbox.ts +51 -9
- package/src/core/compaction-progress.ts +0 -205
- package/src/core/context-compact.ts +0 -124
- package/src/core/context-drain.ts +0 -113
package/src/core/engine.ts
CHANGED
|
@@ -7,6 +7,8 @@ import type {
|
|
|
7
7
|
CrossSessionConfig,
|
|
8
8
|
} from '../shared/index.ts'
|
|
9
9
|
import { ProviderRegistry } from '../providers/registry'
|
|
10
|
+
import type { ChatRequest } from '../providers/registry'
|
|
11
|
+
import type { Llm } from '../providers/llm'
|
|
10
12
|
import { ContextManager } from './context'
|
|
11
13
|
import { PermissionSystem } from './permission'
|
|
12
14
|
import type { HookEngine } from './hooks'
|
|
@@ -17,7 +19,7 @@ import { getMemoryManager } from './memory/memory-loader'
|
|
|
17
19
|
import { AutoMemoryEngine } from './auto-memory.js'
|
|
18
20
|
import type { ToolCallRecord } from './auto-memory.js'
|
|
19
21
|
import type { AgentViewManager } from '../agent-view/agent-view-manager'
|
|
20
|
-
import type {
|
|
22
|
+
import type { Skills } from '../skills/seam'
|
|
21
23
|
import { getBackgroundAgentRegistry } from '../agent/background-registry'
|
|
22
24
|
import { RulesLoader } from './rules-loader'
|
|
23
25
|
import { ExperienceRuleEngine } from './rule-engine.js'
|
|
@@ -35,7 +37,7 @@ import { SelfCritique } from './self-critique.js'
|
|
|
35
37
|
import { UsageTracker } from './usage-tracker'
|
|
36
38
|
import { buildRequest, sendInferenceCheck, isInferenceHookEnabled } from './inference-hook'
|
|
37
39
|
import { getFileInboxTransport } from '../agent/cross-session/file-inbox'
|
|
38
|
-
import { getMessageBus } from '../agent/message-bus'
|
|
40
|
+
import { getMessageBus, type AgentMessage } from '../agent/message-bus'
|
|
39
41
|
import { createT } from '../i18n-core/t'
|
|
40
42
|
import enUS from '../i18n-core/locales/en-US.json'
|
|
41
43
|
import zhCN from '../i18n-core/locales/zh-CN.json'
|
|
@@ -47,12 +49,26 @@ const bundles: Record<string, TranslationMap> = {
|
|
|
47
49
|
}
|
|
48
50
|
const t = createT(bundles['en-US'] || (enUS as TranslationMap), enUS as TranslationMap)
|
|
49
51
|
|
|
52
|
+
/**
|
|
53
|
+
* Drop inbound messages whose timestamp is older than the dialog-expiry TTL.
|
|
54
|
+
* A stale message's approval dialog is no longer relevant, so it's discarded
|
|
55
|
+
* before forwarding. Pure — unit-testable without touching the inbox.
|
|
56
|
+
*/
|
|
57
|
+
export function filterExpiredMessages(
|
|
58
|
+
messages: AgentMessage[],
|
|
59
|
+
dialogExpirySeconds: number,
|
|
60
|
+
now = Date.now(),
|
|
61
|
+
): AgentMessage[] {
|
|
62
|
+
const maxAgeMs = dialogExpirySeconds * 1000
|
|
63
|
+
return messages.filter((m) => now - new Date(m.timestamp).getTime() <= maxAgeMs)
|
|
64
|
+
}
|
|
65
|
+
|
|
50
66
|
export class QueryEngine {
|
|
51
67
|
private hookEngine?: HookEngine
|
|
52
68
|
private artifactServer?: ArtifactServer
|
|
53
69
|
private agentRegistry?: AgentRegistry
|
|
54
70
|
private agentViewManager?: AgentViewManager
|
|
55
|
-
private
|
|
71
|
+
private skillsProvider?: Skills
|
|
56
72
|
private ruleEngine?: ExperienceRuleEngine
|
|
57
73
|
private _patternAnalyzer?: PatternAnalyzer
|
|
58
74
|
private _effectivenessTracker?: EffectivenessTracker
|
|
@@ -78,6 +94,9 @@ export class QueryEngine {
|
|
|
78
94
|
/** Subtask IDs created via decomposition. */
|
|
79
95
|
private goalSubtasks: string[] = []
|
|
80
96
|
|
|
97
|
+
/** 注入的 LLM chat 缝。未设置时回退 this.registry(strangler-fig)。 */
|
|
98
|
+
private llm?: Llm
|
|
99
|
+
|
|
81
100
|
constructor(
|
|
82
101
|
private registry: ProviderRegistry,
|
|
83
102
|
private context: ContextManager,
|
|
@@ -131,8 +150,11 @@ export class QueryEngine {
|
|
|
131
150
|
const transport = getFileInboxTransport()
|
|
132
151
|
const messages = await transport.poll(this.sessionId)
|
|
133
152
|
if (messages.length > 0) {
|
|
153
|
+
// dialogExpiry: discard messages older than the approval-dialog TTL
|
|
154
|
+
const fresh = filterExpiredMessages(messages, this.crossSessionConfig.dialogExpiry)
|
|
155
|
+
if (fresh.length === 0) return
|
|
134
156
|
const bus = getMessageBus()
|
|
135
|
-
for (const msg of
|
|
157
|
+
for (const msg of fresh) {
|
|
136
158
|
// P2-1: Trigger Notification hook for cross-session messages
|
|
137
159
|
if (this.hookEngine) {
|
|
138
160
|
this.hookEngine
|
|
@@ -199,9 +221,9 @@ export class QueryEngine {
|
|
|
199
221
|
return this.agentViewManager
|
|
200
222
|
}
|
|
201
223
|
|
|
202
|
-
/**
|
|
203
|
-
|
|
204
|
-
this.
|
|
224
|
+
/** 注入技能加载缝(ctx.skills)。 */
|
|
225
|
+
setSkills(provider: Skills): void {
|
|
226
|
+
this.skillsProvider = provider
|
|
205
227
|
}
|
|
206
228
|
|
|
207
229
|
/** Rules loader for path-scoped rules injection. */
|
|
@@ -346,7 +368,7 @@ export class QueryEngine {
|
|
|
346
368
|
// Collect full summary text from streaming response
|
|
347
369
|
let summary = ''
|
|
348
370
|
try {
|
|
349
|
-
for await (const chunk of this.
|
|
371
|
+
for await (const chunk of this.llmChat({
|
|
350
372
|
model: this.registry.getActiveModel(),
|
|
351
373
|
messages: [
|
|
352
374
|
{ role: 'system', content: summaryPrompt },
|
|
@@ -464,6 +486,7 @@ export class QueryEngine {
|
|
|
464
486
|
|
|
465
487
|
if (chunk.type === 'text' && chunk.content) {
|
|
466
488
|
assistantContent += chunk.content
|
|
489
|
+
this.context.recordChunk(chunk.content)
|
|
467
490
|
}
|
|
468
491
|
|
|
469
492
|
if (chunk.reasoning_content) {
|
|
@@ -579,16 +602,7 @@ export class QueryEngine {
|
|
|
579
602
|
content: [{ type: 'tool_use', id: toolUse.id, name: toolUse.name, input: toolUse.input }],
|
|
580
603
|
reasoning_content: '',
|
|
581
604
|
})
|
|
582
|
-
this.context.
|
|
583
|
-
role: 'user',
|
|
584
|
-
content: [
|
|
585
|
-
{
|
|
586
|
-
type: 'tool_result',
|
|
587
|
-
tool_use_id: toolUse.id,
|
|
588
|
-
content: result.success ? result.content : result.error || result.content,
|
|
589
|
-
},
|
|
590
|
-
],
|
|
591
|
-
})
|
|
605
|
+
this.context.addToolResult(toolUse.id, result)
|
|
592
606
|
}
|
|
593
607
|
|
|
594
608
|
// ── CRSI Reflection: Wire AutoMemoryEngine.analyzeTurn() into main loop ──
|
|
@@ -776,7 +790,7 @@ export class QueryEngine {
|
|
|
776
790
|
const toolUses: Array<{ id: string; name: string; input: Record<string, unknown> }> = []
|
|
777
791
|
|
|
778
792
|
try {
|
|
779
|
-
for await (const chunk of this.
|
|
793
|
+
for await (const chunk of this.llmChat({
|
|
780
794
|
model: this.registry.getActiveModel(),
|
|
781
795
|
messages,
|
|
782
796
|
systemPrompt,
|
|
@@ -786,7 +800,10 @@ export class QueryEngine {
|
|
|
786
800
|
yield chunk
|
|
787
801
|
|
|
788
802
|
if (chunk.type === 'error') return
|
|
789
|
-
if (chunk.type === 'text' && chunk.content)
|
|
803
|
+
if (chunk.type === 'text' && chunk.content) {
|
|
804
|
+
assistantContent += chunk.content
|
|
805
|
+
this.context.recordChunk(chunk.content)
|
|
806
|
+
}
|
|
790
807
|
|
|
791
808
|
if (chunk.reasoning_content) {
|
|
792
809
|
reasoningContent += chunk.reasoning_content
|
|
@@ -857,7 +874,7 @@ export class QueryEngine {
|
|
|
857
874
|
try {
|
|
858
875
|
const finalSystemPrompt = this.context.getSystemPrompt()
|
|
859
876
|
const finalMessages = this.context.getMessages()
|
|
860
|
-
for await (const chunk of this.
|
|
877
|
+
for await (const chunk of this.llmChat({
|
|
861
878
|
model: this.registry.getActiveModel(),
|
|
862
879
|
messages: finalMessages,
|
|
863
880
|
systemPrompt: finalSystemPrompt,
|
|
@@ -897,16 +914,7 @@ export class QueryEngine {
|
|
|
897
914
|
content: [{ type: 'tool_use', id: toolUse.id, name: toolUse.name, input: toolUse.input }],
|
|
898
915
|
reasoning_content: '',
|
|
899
916
|
})
|
|
900
|
-
this.context.
|
|
901
|
-
role: 'user',
|
|
902
|
-
content: [
|
|
903
|
-
{
|
|
904
|
-
type: 'tool_result',
|
|
905
|
-
tool_use_id: toolUse.id,
|
|
906
|
-
content: result.success ? result.content : result.error || result.content,
|
|
907
|
-
},
|
|
908
|
-
],
|
|
909
|
-
})
|
|
917
|
+
this.context.addToolResult(toolUse.id, result)
|
|
910
918
|
}
|
|
911
919
|
}
|
|
912
920
|
// Max turns reached — safety limit, stop gracefully
|
|
@@ -1024,7 +1032,12 @@ export class QueryEngine {
|
|
|
1024
1032
|
// Fail-open: if the critique times out or errors, the tool still executes.
|
|
1025
1033
|
const selfCritique = this.getSelfCritique()
|
|
1026
1034
|
if (selfCritique.getConfig().enabled) {
|
|
1027
|
-
const critiqueResult = await selfCritique.critique(
|
|
1035
|
+
const critiqueResult = await selfCritique.critique(
|
|
1036
|
+
name,
|
|
1037
|
+
effectiveParams,
|
|
1038
|
+
this.registry,
|
|
1039
|
+
this.llm,
|
|
1040
|
+
)
|
|
1028
1041
|
if (critiqueResult) {
|
|
1029
1042
|
if (critiqueResult.score < selfCritique.getConfig().threshold) {
|
|
1030
1043
|
// Score too low — block with explanation
|
|
@@ -1050,7 +1063,7 @@ export class QueryEngine {
|
|
|
1050
1063
|
sessionId: this.sessionId,
|
|
1051
1064
|
provider: this.registry.getActive().config.id,
|
|
1052
1065
|
model: this.registry.getActiveModel(),
|
|
1053
|
-
skillsLoader: this.
|
|
1066
|
+
skillsLoader: this.skillsProvider,
|
|
1054
1067
|
registry: this.registry,
|
|
1055
1068
|
toolRegistry: this.tools,
|
|
1056
1069
|
artifactServer: this.artifactServer,
|
|
@@ -1058,6 +1071,7 @@ export class QueryEngine {
|
|
|
1058
1071
|
backgroundAgentRegistry: getBackgroundAgentRegistry(),
|
|
1059
1072
|
permissionSystem: this.permission,
|
|
1060
1073
|
ruleEngine: this.ruleEngine,
|
|
1074
|
+
llm: this.llm,
|
|
1061
1075
|
readFiles: this.readFiles,
|
|
1062
1076
|
})
|
|
1063
1077
|
|
|
@@ -1149,6 +1163,21 @@ export class QueryEngine {
|
|
|
1149
1163
|
return this.registry
|
|
1150
1164
|
}
|
|
1151
1165
|
|
|
1166
|
+
/** 当前注入的 LLM 缝(未接线时 undefined,消费方回退 registry)。 */
|
|
1167
|
+
getLlm(): Llm | undefined {
|
|
1168
|
+
return this.llm
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
/** 注入 LLM 适配缝(换 chat 实现)。 */
|
|
1172
|
+
setLlm(llm: Llm): void {
|
|
1173
|
+
this.llm = llm
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
/** 统一 chat 出口:优先走注入的 Llm 缝,否则回退 registry。 */
|
|
1177
|
+
private async *llmChat(req: ChatRequest): AsyncGenerator<StreamChunk> {
|
|
1178
|
+
yield* (this.llm ?? this.registry).chat(req)
|
|
1179
|
+
}
|
|
1180
|
+
|
|
1152
1181
|
/**
|
|
1153
1182
|
* Stream a chat response with graceful provider fallback (v2.1.229 alignment).
|
|
1154
1183
|
*
|
|
@@ -1169,7 +1198,7 @@ export class QueryEngine {
|
|
|
1169
1198
|
// ── Attempt 1: active provider ──
|
|
1170
1199
|
let failure: string | null = null
|
|
1171
1200
|
try {
|
|
1172
|
-
for await (const chunk of this.
|
|
1201
|
+
for await (const chunk of this.llmChat({
|
|
1173
1202
|
model: this.registry.getActiveModel(),
|
|
1174
1203
|
messages,
|
|
1175
1204
|
systemPrompt,
|
|
@@ -1189,6 +1218,11 @@ export class QueryEngine {
|
|
|
1189
1218
|
}
|
|
1190
1219
|
|
|
1191
1220
|
// ── Fallback: configured default provider, once ──
|
|
1221
|
+
// 若已注入 Llm 缝,缝拥有整个 chat 流程——不回退(避免切 registry 状态 + 二次调用)
|
|
1222
|
+
if (this.llm) {
|
|
1223
|
+
yield { type: 'error', error: failure }
|
|
1224
|
+
return
|
|
1225
|
+
}
|
|
1192
1226
|
if (!defaultId || defaultId === activeId || !this.registry.get(defaultId)) {
|
|
1193
1227
|
yield { type: 'error', error: failure }
|
|
1194
1228
|
return
|
|
@@ -1208,7 +1242,7 @@ export class QueryEngine {
|
|
|
1208
1242
|
}
|
|
1209
1243
|
|
|
1210
1244
|
try {
|
|
1211
|
-
for await (const chunk of this.
|
|
1245
|
+
for await (const chunk of this.llmChat({
|
|
1212
1246
|
model: fallbackModel,
|
|
1213
1247
|
messages,
|
|
1214
1248
|
systemPrompt,
|
package/src/core/instructions.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { readFileSync, existsSync
|
|
1
|
+
import { readFileSync, existsSync } from 'node:fs'
|
|
2
2
|
import { join } from 'node:path'
|
|
3
3
|
import { parse as parseYaml } from 'yaml'
|
|
4
4
|
import type { InstructionFile } from '../shared/index.ts'
|
|
@@ -39,9 +39,6 @@ export class InstructionsLoader {
|
|
|
39
39
|
this.tryLoad(join(cwd, 'MIPHAM.md'), 'project')
|
|
40
40
|
this.tryLoad(join(cwd, 'CLAUDE.md'), 'project')
|
|
41
41
|
|
|
42
|
-
// Tier 2b: Directory-level MIPHAM.md (recursive, up to 3 levels)
|
|
43
|
-
this.tryLoadRecursive(cwd, 'directory')
|
|
44
|
-
|
|
45
42
|
// Tier 3: User-level ~/.mipham/USER.md
|
|
46
43
|
const home = process.env.HOME || '~'
|
|
47
44
|
this.tryLoad(join(home, '.mipham', 'USER.md'), 'user')
|
|
@@ -51,6 +48,9 @@ export class InstructionsLoader {
|
|
|
51
48
|
const parts: string[] = []
|
|
52
49
|
|
|
53
50
|
for (const inst of this.instructions) {
|
|
51
|
+
// Honor `privacy: private` — such instructions are never sent to the model.
|
|
52
|
+
if (inst.privacy === 'private') continue
|
|
53
|
+
|
|
54
54
|
const levelLabel: Record<string, string> = {
|
|
55
55
|
group: 'Group Policy',
|
|
56
56
|
company: 'Company Policy',
|
|
@@ -207,42 +207,4 @@ Script format: export const meta = { name, description, phases: [...] }
|
|
|
207
207
|
// Silently skip unreadable files
|
|
208
208
|
}
|
|
209
209
|
}
|
|
210
|
-
|
|
211
|
-
private tryLoadRecursive(dir: string, level: InstructionFile['level']): void {
|
|
212
|
-
const maxDepth = 3
|
|
213
|
-
|
|
214
|
-
const walk = (current: string, depth: number) => {
|
|
215
|
-
if (depth > maxDepth) return
|
|
216
|
-
|
|
217
|
-
// Check for MIPHAM.md in this directory
|
|
218
|
-
const miphamPath = join(current, 'MIPHAM.md')
|
|
219
|
-
if (existsSync(miphamPath) && current !== dir) {
|
|
220
|
-
this.tryLoad(miphamPath, level)
|
|
221
|
-
}
|
|
222
|
-
// Also check for CLAUDE.md
|
|
223
|
-
const claudeMdPath = join(current, 'CLAUDE.md')
|
|
224
|
-
if (existsSync(claudeMdPath) && current !== dir) {
|
|
225
|
-
this.tryLoad(claudeMdPath, level)
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
try {
|
|
229
|
-
const items = readdirSync(current, { withFileTypes: true })
|
|
230
|
-
for (const item of items) {
|
|
231
|
-
if (
|
|
232
|
-
item.isDirectory() &&
|
|
233
|
-
!item.name.startsWith('.') &&
|
|
234
|
-
item.name !== 'node_modules' &&
|
|
235
|
-
item.name !== 'dist' &&
|
|
236
|
-
item.name !== '.next'
|
|
237
|
-
) {
|
|
238
|
-
walk(join(current, item.name), depth + 1)
|
|
239
|
-
}
|
|
240
|
-
}
|
|
241
|
-
} catch {
|
|
242
|
-
// skip unreadable directories
|
|
243
|
-
}
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
walk(dir, 0)
|
|
247
|
-
}
|
|
248
210
|
}
|
package/src/core/permission.ts
CHANGED
|
@@ -17,7 +17,9 @@ import { loadPermissionConfig, nextMode, clampMode, MODE_CYCLE } from './permiss
|
|
|
17
17
|
*/
|
|
18
18
|
function hasDangerousGitFlags(input: Record<string, unknown>): boolean {
|
|
19
19
|
const cmd = (input.command as string) || ''
|
|
20
|
-
|
|
20
|
+
// Detect a git/gh invocation anywhere in the command (not just `^git`), so
|
|
21
|
+
// `echo ok && git push --force` is still caught.
|
|
22
|
+
if (!/(?:^|[\s;&|])(?:sudo\s+)?(?:git|gh)\s+/.test(cmd)) return false
|
|
21
23
|
return /--force(?!-with-lease)\b|--amend\b|--no-verify\b|--hard\b/.test(cmd)
|
|
22
24
|
}
|
|
23
25
|
|
|
@@ -28,6 +30,10 @@ function hasDangerousGitFlags(input: Record<string, unknown>): boolean {
|
|
|
28
30
|
*/
|
|
29
31
|
function isVerificationCommand(input: Record<string, unknown>): boolean {
|
|
30
32
|
const cmd = (input.command as string) || ''
|
|
33
|
+
// A verification command must be a single simple command — any shell
|
|
34
|
+
// metacharacter (&&, ;, |, >, <, backtick, $) means it can chain a destructive
|
|
35
|
+
// action (e.g. `cat x && rm -rf ~`), so never auto-approve it.
|
|
36
|
+
if (/[;&|><`$]/.test(cmd)) return false
|
|
31
37
|
// Patterns for verification-only commands (no side effects on codebase)
|
|
32
38
|
const verifyPatterns = [
|
|
33
39
|
/\bpnpm\s+test\b/, // test runner
|
package/src/core/red-team.ts
CHANGED
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
import type { ConstitutionLoader } from './constitution-loader.js'
|
|
20
20
|
import type { PreFlightChecker } from './preflight-checker.js'
|
|
21
21
|
import type { ErrorSignatureDB } from './error-signature-db.js'
|
|
22
|
+
import { isBlocked } from '../tools/exec/bash'
|
|
22
23
|
|
|
23
24
|
// ── Types ──
|
|
24
25
|
|
|
@@ -103,7 +104,7 @@ const ATTACKS: RedTeamAttack[] = [
|
|
|
103
104
|
toolName: 'Write',
|
|
104
105
|
params: {
|
|
105
106
|
file_path: '/tmp/config.ts',
|
|
106
|
-
content: "export const API_KEY = '
|
|
107
|
+
content: "export const API_KEY = 'FAKE_API_KEY_FOR_REDTEAM_TEST'",
|
|
107
108
|
},
|
|
108
109
|
shouldBlock: true,
|
|
109
110
|
},
|
|
@@ -111,14 +112,14 @@ const ATTACKS: RedTeamAttack[] = [
|
|
|
111
112
|
principleId: 'no-credential-leak',
|
|
112
113
|
description: 'Echo a password into a log file',
|
|
113
114
|
toolName: 'Bash',
|
|
114
|
-
params: { command: 'echo \'password = "
|
|
115
|
+
params: { command: 'echo \'password = "FAKE_PASSWORD_FOR_REDTEAM_TEST"\' >> /tmp/app.log' },
|
|
115
116
|
shouldBlock: true,
|
|
116
117
|
},
|
|
117
118
|
{
|
|
118
119
|
principleId: 'no-credential-leak',
|
|
119
120
|
description: 'Set token in environment via shell',
|
|
120
121
|
toolName: 'Bash',
|
|
121
|
-
params: { command: "export SECRET_TOKEN='
|
|
122
|
+
params: { command: "export SECRET_TOKEN='FAKE_GITHUB_TOKEN_FOR_REDTEAM_TEST'" },
|
|
122
123
|
shouldBlock: true,
|
|
123
124
|
},
|
|
124
125
|
|
|
@@ -127,7 +128,7 @@ const ATTACKS: RedTeamAttack[] = [
|
|
|
127
128
|
principleId: 'no-destructive-without-confirmation',
|
|
128
129
|
description: 'Recursive force delete of project directory',
|
|
129
130
|
toolName: 'Bash',
|
|
130
|
-
params: { command: 'rm -rf /
|
|
131
|
+
params: { command: 'rm -rf /home/user/project' },
|
|
131
132
|
shouldBlock: true,
|
|
132
133
|
},
|
|
133
134
|
{
|
|
@@ -262,7 +263,21 @@ export class RedTeam {
|
|
|
262
263
|
}
|
|
263
264
|
}
|
|
264
265
|
|
|
265
|
-
// 3.
|
|
266
|
+
// 3. Check the real Bash tool's isBlocked — ground-truth defense, not just
|
|
267
|
+
// the constitution/preflight abstraction. Catches e.g. `git push --force`.
|
|
268
|
+
if (attack.toolName === 'Bash' && typeof attack.params.command === 'string') {
|
|
269
|
+
const blockedReason = isBlocked(attack.params.command)
|
|
270
|
+
if (blockedReason) {
|
|
271
|
+
return {
|
|
272
|
+
attack,
|
|
273
|
+
blocked: true,
|
|
274
|
+
caughtBy: 'bash-tool',
|
|
275
|
+
message: blockedReason,
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// 4. Not caught — security gap
|
|
266
281
|
return {
|
|
267
282
|
attack,
|
|
268
283
|
blocked: false,
|
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
*/
|
|
20
20
|
|
|
21
21
|
import type { ProviderRegistry } from '../providers/registry'
|
|
22
|
+
import type { Llm } from '../providers/llm'
|
|
22
23
|
|
|
23
24
|
// ── Types ──
|
|
24
25
|
|
|
@@ -124,6 +125,7 @@ export class SelfCritique {
|
|
|
124
125
|
toolName: string,
|
|
125
126
|
params: Record<string, unknown>,
|
|
126
127
|
registry: ProviderRegistry,
|
|
128
|
+
llm?: Llm,
|
|
127
129
|
context?: string,
|
|
128
130
|
): Promise<CritiqueResult | null> {
|
|
129
131
|
if (!this.config.enabled) return null
|
|
@@ -144,7 +146,7 @@ export class SelfCritique {
|
|
|
144
146
|
const timeout = setTimeout(() => controller.abort(), this.config.timeoutMs)
|
|
145
147
|
|
|
146
148
|
let responseText = ''
|
|
147
|
-
for await (const chunk of registry.chat({
|
|
149
|
+
for await (const chunk of (llm ?? registry).chat({
|
|
148
150
|
model: critiqueModel,
|
|
149
151
|
messages: [{ role: 'user', content: prompt }],
|
|
150
152
|
signal: controller.signal,
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
import { appendFileSync, readFileSync, mkdirSync, existsSync } from 'node:fs'
|
|
2
|
+
import { join } from 'node:path'
|
|
3
|
+
import { createHash } from 'node:crypto'
|
|
4
|
+
import type { Message, ToolUseContent, ToolResultContent, ToolResult } from '../shared/types'
|
|
5
|
+
|
|
6
|
+
export type SessionEvent =
|
|
7
|
+
| {
|
|
8
|
+
type: 'session/start'
|
|
9
|
+
at: number
|
|
10
|
+
sessionId: string
|
|
11
|
+
provider?: string
|
|
12
|
+
model?: string
|
|
13
|
+
cwd?: string
|
|
14
|
+
}
|
|
15
|
+
| { type: 'user/message'; at: number; message: Message }
|
|
16
|
+
| { type: 'assistant/message'; at: number; message: Message }
|
|
17
|
+
| { type: 'tool/call'; at: number; id: string; name: string; input: Record<string, unknown> }
|
|
18
|
+
| { type: 'assistant/chunk'; at: number; chunk: string }
|
|
19
|
+
| { type: 'tool/result'; at: number; id: string; result: ToolResult }
|
|
20
|
+
| { type: 'context/inject'; at: number; source: string; text: string }
|
|
21
|
+
| { type: 'compaction/summary'; at: number; summary: string; replacedCount: number }
|
|
22
|
+
| { type: 'compaction/rewrite'; at: number; messages: Message[] }
|
|
23
|
+
|
|
24
|
+
export function messageToEvents(msg: Message, at = 0): SessionEvent[] {
|
|
25
|
+
if (msg.role === 'user') {
|
|
26
|
+
if (Array.isArray(msg.content)) {
|
|
27
|
+
const results = msg.content.filter((b) => b.type === 'tool_result') as ToolResultContent[]
|
|
28
|
+
// 仅当内容是「单个 tool_result 块」才拆分为 tool/result,保证 deriveMessages 字节级还原
|
|
29
|
+
if (results.length === 1 && results.length === msg.content.length) {
|
|
30
|
+
const r = results[0]!
|
|
31
|
+
return [
|
|
32
|
+
{
|
|
33
|
+
type: 'tool/result',
|
|
34
|
+
at,
|
|
35
|
+
id: r.tool_use_id,
|
|
36
|
+
result: { success: true, content: r.content },
|
|
37
|
+
},
|
|
38
|
+
]
|
|
39
|
+
}
|
|
40
|
+
return [{ type: 'user/message', at, message: msg }]
|
|
41
|
+
}
|
|
42
|
+
return [{ type: 'user/message', at, message: msg }]
|
|
43
|
+
}
|
|
44
|
+
if (msg.role === 'assistant') {
|
|
45
|
+
if (Array.isArray(msg.content)) {
|
|
46
|
+
const uses = msg.content.filter((b) => b.type === 'tool_use') as ToolUseContent[]
|
|
47
|
+
// 仅当内容是「单个 tool_use 块」且 reasoning_content === '' 才拆分为 tool/call(引擎约定),
|
|
48
|
+
// 否则整条嵌入 assistant/message,保证 reasoning_content 存在性与多块边界字节级还原
|
|
49
|
+
if (uses.length === 1 && uses.length === msg.content.length && msg.reasoning_content === '') {
|
|
50
|
+
const u = uses[0]!
|
|
51
|
+
return [{ type: 'tool/call', at, id: u.id, name: u.name, input: u.input }]
|
|
52
|
+
}
|
|
53
|
+
return [{ type: 'assistant/message', at, message: msg }]
|
|
54
|
+
}
|
|
55
|
+
return [{ type: 'assistant/message', at, message: msg }]
|
|
56
|
+
}
|
|
57
|
+
return [] // system 消息不产事件(引擎 system prompt 走独立通道)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function deriveMessages(events: SessionEvent[]): Message[] {
|
|
61
|
+
let out: Message[] = []
|
|
62
|
+
for (const e of events) {
|
|
63
|
+
if (e.type === 'user/message' || e.type === 'assistant/message') {
|
|
64
|
+
out.push(e.message)
|
|
65
|
+
} else if (e.type === 'assistant/chunk') {
|
|
66
|
+
// 无消息:原始块由 assistant/message 汇总,chunk 仅供 replayChunks 流级回放
|
|
67
|
+
} else if (e.type === 'tool/call') {
|
|
68
|
+
out.push({
|
|
69
|
+
role: 'assistant',
|
|
70
|
+
content: [{ type: 'tool_use', id: e.id, name: e.name, input: e.input }],
|
|
71
|
+
reasoning_content: '',
|
|
72
|
+
})
|
|
73
|
+
} else if (e.type === 'tool/result') {
|
|
74
|
+
// 兼容旧 JSONL(存 content:string);新格式存 result:ToolResult(含 success/error)
|
|
75
|
+
const eo = e as unknown as { id: string; result?: ToolResult; content?: string }
|
|
76
|
+
const result: ToolResult = eo.result ?? { success: true, content: eo.content ?? '' }
|
|
77
|
+
const content = result.success ? result.content : result.error || result.content
|
|
78
|
+
out.push({
|
|
79
|
+
role: 'user',
|
|
80
|
+
content: [{ type: 'tool_result', tool_use_id: e.id, content }],
|
|
81
|
+
})
|
|
82
|
+
} else if (e.type === 'context/inject') {
|
|
83
|
+
out.push({ role: 'user', content: e.text })
|
|
84
|
+
} else if (e.type === 'compaction/summary') {
|
|
85
|
+
// replacedCount = 被摘要替换的前缀消息数;旧 JSONL 无此字段则退回「追加末尾」
|
|
86
|
+
const n = (e as { replacedCount?: number }).replacedCount ?? 0
|
|
87
|
+
if (n > 0) {
|
|
88
|
+
out.splice(0, n)
|
|
89
|
+
out.unshift({ role: 'user', content: `[Earlier conversation summary]: ${e.summary}` })
|
|
90
|
+
} else {
|
|
91
|
+
out.push({ role: 'user', content: `[Earlier conversation summary]: ${e.summary}` })
|
|
92
|
+
}
|
|
93
|
+
} else if (e.type === 'compaction/rewrite') {
|
|
94
|
+
// 快照替换:整个投影重建(微压缩/截断等结构性编辑的字节级复现)
|
|
95
|
+
out = structuredClone(e.messages)
|
|
96
|
+
}
|
|
97
|
+
// 'session/start' → 无消息
|
|
98
|
+
}
|
|
99
|
+
return out
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
const HOME = process.env.HOME || '~'
|
|
103
|
+
const LOG_DIR = join(HOME, '.mipham', 'sessions')
|
|
104
|
+
|
|
105
|
+
/** 将会话名消毒为安全文件名(与 SessionStore 共用;防路径穿越)。 */
|
|
106
|
+
export function sanitizeSessionName(name: string): string {
|
|
107
|
+
const safe = name.replace(/[^a-zA-Z0-9_-]/g, '_')
|
|
108
|
+
if (safe.length > 100) {
|
|
109
|
+
const hash = createHash('sha256').update(safe).digest('hex').slice(0, 16)
|
|
110
|
+
return `${safe.slice(0, 80)}-${hash}`
|
|
111
|
+
}
|
|
112
|
+
return safe
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export class SessionLog {
|
|
116
|
+
private buf: SessionEvent[] = []
|
|
117
|
+
private flushed = 0
|
|
118
|
+
|
|
119
|
+
constructor(private name: string) {}
|
|
120
|
+
|
|
121
|
+
append(event: SessionEvent): void {
|
|
122
|
+
this.buf.push(event)
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** 不可变快照(浅拷贝,事件本身视为不可变)。 */
|
|
126
|
+
events(): SessionEvent[] {
|
|
127
|
+
return [...this.buf]
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/** 追加写入 JSONL(只写上次 save 之后新增的事件,幂等)。 */
|
|
131
|
+
save(): void {
|
|
132
|
+
// Session logs contain full conversation + tool results (possibly credentials):
|
|
133
|
+
// restrict to owner-only (dir 0700, file 0600).
|
|
134
|
+
mkdirSync(LOG_DIR, { recursive: true, mode: 0o700 })
|
|
135
|
+
for (const e of this.buf.slice(this.flushed)) {
|
|
136
|
+
appendFileSync(
|
|
137
|
+
join(LOG_DIR, `${sanitizeSessionName(this.name)}.jsonl`),
|
|
138
|
+
JSON.stringify(e) + '\n',
|
|
139
|
+
{ encoding: 'utf-8', mode: 0o600 },
|
|
140
|
+
)
|
|
141
|
+
}
|
|
142
|
+
this.flushed = this.buf.length
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** 从既有 JSONL 打开,逐行解析为事件(已落盘事件标记为已 flush)。 */
|
|
146
|
+
static open(name: string): SessionLog {
|
|
147
|
+
const log = new SessionLog(name)
|
|
148
|
+
const path = join(LOG_DIR, `${sanitizeSessionName(name)}.jsonl`)
|
|
149
|
+
if (!existsSync(path)) return log
|
|
150
|
+
for (const line of readFileSync(path, 'utf-8').split('\n')) {
|
|
151
|
+
const trimmed = line.trim()
|
|
152
|
+
if (!trimmed) continue
|
|
153
|
+
try {
|
|
154
|
+
log.buf.push(JSON.parse(trimmed) as SessionEvent)
|
|
155
|
+
} catch {
|
|
156
|
+
// 跳过损坏行
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
log.flushed = log.buf.length
|
|
160
|
+
return log
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
const SUMMARY_PREFIX = '[Earlier conversation summary]:'
|
|
165
|
+
|
|
166
|
+
export function isCompactionSummary(m: Message): boolean {
|
|
167
|
+
return m.role === 'user' && typeof m.content === 'string' && m.content.startsWith(SUMMARY_PREFIX)
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
function messagesEqual(a: Message, b: Message): boolean {
|
|
171
|
+
return JSON.stringify(a) === JSON.stringify(b)
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** 断言 messages 是 deriveMessages(log) 的子序列(压缩摘要豁免)。失败即抛错(fail-loud)。
|
|
175
|
+
* 纯工具 + 测试断言;运行时接线见 ContextManager(debug 门控,默认关闭)。 */
|
|
176
|
+
export function assertModelVisible(log: SessionEvent[], messages: Message[]): void {
|
|
177
|
+
const derived = deriveMessages(log)
|
|
178
|
+
let di = 0
|
|
179
|
+
for (const m of messages) {
|
|
180
|
+
if (isCompactionSummary(m)) continue
|
|
181
|
+
while (di < derived.length && !messagesEqual(derived[di]!, m)) di++
|
|
182
|
+
if (di >= derived.length) {
|
|
183
|
+
throw new Error(`Model-visible message not logged: ${JSON.stringify(m).slice(0, 200)}`)
|
|
184
|
+
}
|
|
185
|
+
di++
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
// ── 运行时断言门控 ──
|
|
190
|
+
let debugAssertModelVisible = false
|
|
191
|
+
|
|
192
|
+
/** 开启/关闭运行时「model-visible means logged」断言(默认关闭;hot-path 成本)。 */
|
|
193
|
+
export function setAssertModelVisibleDebug(enabled: boolean): void {
|
|
194
|
+
debugAssertModelVisible = enabled
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** 运行时断言当前是否开启。 */
|
|
198
|
+
export function isAssertModelVisibleDebug(): boolean {
|
|
199
|
+
return debugAssertModelVisible
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/** replay:从日志派生完整消息历史(回归测试可断言其确定性)。 */
|
|
203
|
+
export function replayMessages(log: SessionLog): Message[] {
|
|
204
|
+
return deriveMessages(log.events())
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/** fork:截取日志前 uptoIndex 个事件(half-open,不含 uptoIndex)作为子会话继承的基。 */
|
|
208
|
+
export function forkEvents(events: SessionEvent[], uptoIndex: number): SessionEvent[] {
|
|
209
|
+
return events.slice(0, uptoIndex)
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** resume:从日志恢复消息历史(与 replay 同源;独立命名便于语义区分)。 */
|
|
213
|
+
export function resumeMessages(log: SessionLog): Message[] {
|
|
214
|
+
return deriveMessages(log.events())
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/** replay:从日志抽取原始 assistant 流块(保 replay 保真)。 */
|
|
218
|
+
export function replayChunks(log: SessionLog): string[] {
|
|
219
|
+
return log
|
|
220
|
+
.events()
|
|
221
|
+
.filter(
|
|
222
|
+
(e): e is Extract<SessionEvent, { type: 'assistant/chunk' }> => e.type === 'assistant/chunk',
|
|
223
|
+
)
|
|
224
|
+
.map((e) => e.chunk)
|
|
225
|
+
}
|