@miphamai/cli 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/workflows/audit.js +51 -0
- package/skills/workflows/hunt.js +95 -0
- package/skills/workflows/judge.js +55 -0
- package/skills/workflows/migrate.js +92 -0
- package/skills/workflows/research.js +87 -0
- package/skills/workflows/review.js +92 -0
- package/src/agent/background-registry.ts +11 -0
- package/src/agent-view/agent-view-manager.ts +10 -0
- package/src/config/defaults.ts +26 -1
- package/src/config/loader.ts +79 -1
- package/src/core/credential-masker.ts +203 -0
- package/src/core/instructions.ts +28 -0
- package/src/core/permission.ts +62 -8
- package/src/index.tsx +8 -1
- package/src/plugin/plugin-manager.ts +39 -2
- package/src/shared/types.ts +43 -0
- package/src/tools/agent/workflow.ts +67 -3
- package/src/tools/exec/bash.ts +29 -3
- package/src/tools/file/read.ts +30 -1
- package/src/ui/app.tsx +6 -1
- package/src/ui/chat.tsx +50 -17
- package/src/ui/commands.ts +451 -0
- package/src/workflow/primitives/loop.ts +103 -0
- package/src/workflow/primitives/verify.ts +275 -0
- package/src/workflow/runtime.ts +17 -1
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import { parallel } from './parallel'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* loopUntilConvergence() — workflow primitive for iterative discovery with
|
|
5
|
+
* deduplication, adversarial verification, and convergence detection.
|
|
6
|
+
*
|
|
7
|
+
* Orchestrates multiple "finders" over successive rounds until no new items
|
|
8
|
+
* are discovered for `dryRounds` consecutive rounds (convergence), or
|
|
9
|
+
* `maxRounds` is reached (cutoff).
|
|
10
|
+
*
|
|
11
|
+
* Each item is keyed by `keyFn` for deduplication across rounds. If a
|
|
12
|
+
* `verify` function is provided, only items that survive verification
|
|
13
|
+
* enter the confirmed set — but all seen items (even those that fail
|
|
14
|
+
* verification) are tracked in the seen-set to prevent re-discovery.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
export interface VerifyVote {
|
|
18
|
+
real: boolean
|
|
19
|
+
reason: string
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export interface VerifyResult<T = unknown> {
|
|
23
|
+
finding: T
|
|
24
|
+
survives: boolean
|
|
25
|
+
votes: VerifyVote[]
|
|
26
|
+
score: number
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export interface LoopUntilConvergenceResult<T> {
|
|
30
|
+
confirmed: T[]
|
|
31
|
+
totalSeen: number
|
|
32
|
+
rounds: number
|
|
33
|
+
converged: boolean
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface LoopUntilConvergenceOpts<T> {
|
|
37
|
+
finders: Array<() => Promise<{ items: T[] } | null>>
|
|
38
|
+
keyFn: (item: T) => string
|
|
39
|
+
verify?: (item: T) => Promise<VerifyResult<T>>
|
|
40
|
+
dryRounds?: number
|
|
41
|
+
maxRounds?: number
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export async function loopUntilConvergence<T>(
|
|
45
|
+
opts: LoopUntilConvergenceOpts<T>,
|
|
46
|
+
): Promise<LoopUntilConvergenceResult<T>> {
|
|
47
|
+
const dryRounds = opts.dryRounds ?? 2
|
|
48
|
+
const maxRounds = opts.maxRounds ?? 20
|
|
49
|
+
|
|
50
|
+
const seen = new Set<string>()
|
|
51
|
+
const confirmed: T[] = []
|
|
52
|
+
let dry = 0
|
|
53
|
+
let rounds = 0
|
|
54
|
+
|
|
55
|
+
while (dry < dryRounds && rounds < maxRounds) {
|
|
56
|
+
rounds++
|
|
57
|
+
|
|
58
|
+
// FAN OUT: all finders run in parallel
|
|
59
|
+
const raw = await parallel(opts.finders.map((f) => () => f()))
|
|
60
|
+
|
|
61
|
+
// EDGE LOGIC: flatMap + dedup (pure JS, zero tokens)
|
|
62
|
+
const items: T[] = []
|
|
63
|
+
for (const result of raw) {
|
|
64
|
+
if (result && result.items) {
|
|
65
|
+
items.push(...result.items)
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// Dedup against SEEN set, not confirmed
|
|
70
|
+
const fresh = items.filter((item) => {
|
|
71
|
+
const key = opts.keyFn(item)
|
|
72
|
+
if (seen.has(key)) return false
|
|
73
|
+
seen.add(key)
|
|
74
|
+
return true
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
if (fresh.length === 0) {
|
|
78
|
+
dry++ // no new unique items → trending toward convergence
|
|
79
|
+
continue
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
dry = 0 // new items found → reset dry counter
|
|
83
|
+
|
|
84
|
+
// VERIFY: optional quality gate
|
|
85
|
+
if (opts.verify) {
|
|
86
|
+
const judged = await parallel(fresh.map((item) => () => opts.verify!(item)))
|
|
87
|
+
for (const j of judged) {
|
|
88
|
+
if (j && j.survives) {
|
|
89
|
+
confirmed.push(j.finding as T)
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
} else {
|
|
93
|
+
confirmed.push(...fresh)
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
confirmed,
|
|
99
|
+
totalSeen: seen.size,
|
|
100
|
+
rounds,
|
|
101
|
+
converged: dry >= dryRounds,
|
|
102
|
+
}
|
|
103
|
+
}
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* verify() and judge() — workflow primitives for adversarial verification,
|
|
3
|
+
* multi-perspective review, and multi-judge evaluation.
|
|
4
|
+
*
|
|
5
|
+
* verify() supports 3 modes:
|
|
6
|
+
* - adversarial: N skeptics try to refute the finding (majority wins)
|
|
7
|
+
* - perspective: N lenses each evaluate from a specific angle (at least 1 confirms)
|
|
8
|
+
* - consensus: N voters must unanimously agree
|
|
9
|
+
*
|
|
10
|
+
* judge() evaluates N attempts by M judges, computes average scores,
|
|
11
|
+
* picks the winner, and optionally synthesizes a final result.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { workflowAgent } from './agent'
|
|
15
|
+
import { parallel } from './parallel'
|
|
16
|
+
import type { WorkflowAgentOpts } from './agent'
|
|
17
|
+
|
|
18
|
+
// Agent function signature matching the sandbox-injected pattern:
|
|
19
|
+
// (prompt, opts?) → result. In production, the workflow runtime binds
|
|
20
|
+
// ProviderRegistry + ToolRegistry into a function of this shape and
|
|
21
|
+
// injects it via _mockAgent.
|
|
22
|
+
type AgentFn = (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
23
|
+
|
|
24
|
+
// ── Types ──
|
|
25
|
+
|
|
26
|
+
export interface VerifyResult {
|
|
27
|
+
finding: unknown
|
|
28
|
+
survives: boolean
|
|
29
|
+
votes: Array<{ real: boolean; reason: string; lens?: string }>
|
|
30
|
+
score: number
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export type VerifyMode = 'adversarial' | 'perspective' | 'consensus'
|
|
34
|
+
|
|
35
|
+
export interface VerifyOpts {
|
|
36
|
+
mode: VerifyMode
|
|
37
|
+
skeptics?: number // adversarial: default 3
|
|
38
|
+
lenses?: string[] // perspective: e.g. ['correctness', 'security']
|
|
39
|
+
voters?: number // consensus: default 3
|
|
40
|
+
threshold?: number
|
|
41
|
+
schema: Record<string, unknown>
|
|
42
|
+
/** Test-only: inject a mock agent function. When not provided, falls back to workflowAgent. */
|
|
43
|
+
_mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
interface VerdictVote {
|
|
47
|
+
real: boolean
|
|
48
|
+
reason: string
|
|
49
|
+
lens?: string
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export interface JudgeResult {
|
|
53
|
+
winner: unknown
|
|
54
|
+
winnerIndex: number
|
|
55
|
+
scores: Array<{
|
|
56
|
+
attemptIndex: number
|
|
57
|
+
judgeIndex: number
|
|
58
|
+
criteria: Record<string, number>
|
|
59
|
+
total: number
|
|
60
|
+
notes: string
|
|
61
|
+
}>
|
|
62
|
+
synthesis?: string
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export interface JudgeOpts {
|
|
66
|
+
criteria: string[]
|
|
67
|
+
judges?: number // default: 3
|
|
68
|
+
synthesize?: boolean // default: true
|
|
69
|
+
schema: Record<string, unknown>
|
|
70
|
+
/** Test-only: inject a mock agent function. */
|
|
71
|
+
_mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ── Helpers ──
|
|
75
|
+
|
|
76
|
+
function defaultThreshold(mode: VerifyMode, total: number): number {
|
|
77
|
+
if (mode === 'consensus') return total // all must agree
|
|
78
|
+
if (mode === 'perspective') return 1 // at least one lens confirms
|
|
79
|
+
return Math.ceil(total / 2) // adversarial: majority
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// ── verify() ──
|
|
83
|
+
|
|
84
|
+
export async function verify(finding: unknown, opts: VerifyOpts): Promise<VerifyResult> {
|
|
85
|
+
// Prefer the test hook (_mockAgent), fall back to workflowAgent.
|
|
86
|
+
// In production sandbox usage, _mockAgent is set to the pre-bound agent
|
|
87
|
+
// function injected by the workflow runtime.
|
|
88
|
+
const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
|
|
89
|
+
const mode = opts.mode
|
|
90
|
+
|
|
91
|
+
const findingStr = JSON.stringify(finding, null, 2)
|
|
92
|
+
const schemaDesc = JSON.stringify(opts.schema)
|
|
93
|
+
|
|
94
|
+
let prompts: Array<{ prompt: string; lens?: string }>
|
|
95
|
+
|
|
96
|
+
switch (mode) {
|
|
97
|
+
case 'adversarial': {
|
|
98
|
+
const count = opts.skeptics ?? 3
|
|
99
|
+
prompts = Array.from({ length: count }, (_, i) => ({
|
|
100
|
+
prompt:
|
|
101
|
+
`You are a skeptical reviewer (skeptic #${i + 1}). Try to REFUTE this finding. Default to real=false if uncertain.\n\n` +
|
|
102
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
103
|
+
}))
|
|
104
|
+
break
|
|
105
|
+
}
|
|
106
|
+
case 'perspective': {
|
|
107
|
+
const lenses = opts.lenses ?? ['correctness']
|
|
108
|
+
prompts = lenses.map((lens) => ({
|
|
109
|
+
prompt:
|
|
110
|
+
`Judge this finding through the "${lens}" lens. Is it valid from this perspective?\n\n` +
|
|
111
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
112
|
+
lens,
|
|
113
|
+
}))
|
|
114
|
+
break
|
|
115
|
+
}
|
|
116
|
+
case 'consensus': {
|
|
117
|
+
const count = opts.voters ?? 3
|
|
118
|
+
prompts = Array.from({ length: count }, () => ({
|
|
119
|
+
prompt:
|
|
120
|
+
`Is this finding correct? Be honest and critical. Vote real=true only if you are fully convinced.\n\n` +
|
|
121
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
122
|
+
}))
|
|
123
|
+
break
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// Fan out all agent calls concurrently
|
|
128
|
+
const rawVotes = await parallel(
|
|
129
|
+
prompts.map(
|
|
130
|
+
(p) => () => agentFn(p.prompt, { schema: opts.schema as WorkflowAgentOpts['schema'] }),
|
|
131
|
+
),
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
// Filter out failed agents (null) and build vote objects
|
|
135
|
+
const votes: VerdictVote[] = []
|
|
136
|
+
for (let i = 0; i < rawVotes.length; i++) {
|
|
137
|
+
const v = rawVotes[i]
|
|
138
|
+
if (v && typeof v === 'object') {
|
|
139
|
+
const vote = v as Record<string, unknown>
|
|
140
|
+
votes.push({
|
|
141
|
+
real: Boolean(vote.real),
|
|
142
|
+
reason: String(vote.reason ?? ''),
|
|
143
|
+
lens: prompts[i]?.lens,
|
|
144
|
+
})
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const threshold = opts.threshold ?? defaultThreshold(mode, votes.length)
|
|
149
|
+
const realCount = votes.filter((v) => v.real).length
|
|
150
|
+
const survives = realCount >= threshold
|
|
151
|
+
|
|
152
|
+
return {
|
|
153
|
+
finding,
|
|
154
|
+
survives,
|
|
155
|
+
votes,
|
|
156
|
+
score: votes.length > 0 ? realCount / votes.length : 0,
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// ── judge() ──
|
|
161
|
+
|
|
162
|
+
export async function judge(attempts: unknown[], opts: JudgeOpts): Promise<JudgeResult> {
|
|
163
|
+
// Prefer the test hook, fall back to workflowAgent.
|
|
164
|
+
const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
|
|
165
|
+
const judgeCount = opts.judges ?? 3
|
|
166
|
+
const schemaDesc = JSON.stringify(opts.schema)
|
|
167
|
+
|
|
168
|
+
// Phase 1: each judge scores each attempt
|
|
169
|
+
interface ScoreEntry {
|
|
170
|
+
attemptIndex: number
|
|
171
|
+
judgeIndex: number
|
|
172
|
+
criteria: Record<string, number>
|
|
173
|
+
total: number
|
|
174
|
+
notes: string
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// Build the cross-product of judges × attempts
|
|
178
|
+
const scorePrompts: Array<{
|
|
179
|
+
attempt: unknown
|
|
180
|
+
attemptIndex: number
|
|
181
|
+
judgeIndex: number
|
|
182
|
+
}> = []
|
|
183
|
+
for (let ji = 0; ji < judgeCount; ji++) {
|
|
184
|
+
for (let ai = 0; ai < attempts.length; ai++) {
|
|
185
|
+
scorePrompts.push({
|
|
186
|
+
attempt: attempts[ai],
|
|
187
|
+
attemptIndex: ai,
|
|
188
|
+
judgeIndex: ji,
|
|
189
|
+
})
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// Fan out all scoring calls concurrently
|
|
194
|
+
const rawScores = await parallel(
|
|
195
|
+
scorePrompts.map(
|
|
196
|
+
(sp) => () =>
|
|
197
|
+
agentFn(
|
|
198
|
+
`You are judge #${sp.judgeIndex + 1}. Score this attempt against the criteria: ${opts.criteria.join(', ')}.\n\n` +
|
|
199
|
+
`Attempt:\n${JSON.stringify(sp.attempt, null, 2)}\n\n` +
|
|
200
|
+
`Return JSON matching this schema:\n${schemaDesc}`,
|
|
201
|
+
{ schema: opts.schema as WorkflowAgentOpts['schema'] },
|
|
202
|
+
),
|
|
203
|
+
),
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
// Parse and validate each score result
|
|
207
|
+
const scores: ScoreEntry[] = []
|
|
208
|
+
for (let i = 0; i < rawScores.length; i++) {
|
|
209
|
+
const raw = rawScores[i]
|
|
210
|
+
const sp = scorePrompts[i]!
|
|
211
|
+
if (raw && typeof raw === 'object') {
|
|
212
|
+
const obj = raw as Record<string, unknown>
|
|
213
|
+
const criteriaObj = (obj.scores as Record<string, number>) ?? {}
|
|
214
|
+
const total = Object.values(criteriaObj).reduce(
|
|
215
|
+
(sum, v) => sum + (typeof v === 'number' ? v : 0),
|
|
216
|
+
0,
|
|
217
|
+
)
|
|
218
|
+
scores.push({
|
|
219
|
+
attemptIndex: sp.attemptIndex,
|
|
220
|
+
judgeIndex: sp.judgeIndex,
|
|
221
|
+
criteria: criteriaObj,
|
|
222
|
+
total,
|
|
223
|
+
notes: String(obj.notes ?? ''),
|
|
224
|
+
})
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// Compute winner: highest average total across judges
|
|
229
|
+
const attemptTotals = new Map<number, number>()
|
|
230
|
+
const attemptCounts = new Map<number, number>()
|
|
231
|
+
for (const s of scores) {
|
|
232
|
+
attemptTotals.set(s.attemptIndex, (attemptTotals.get(s.attemptIndex) ?? 0) + s.total)
|
|
233
|
+
attemptCounts.set(s.attemptIndex, (attemptCounts.get(s.attemptIndex) ?? 0) + 1)
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
let winnerIndex = 0
|
|
237
|
+
let bestAvg = -Infinity
|
|
238
|
+
for (const [idx, total] of attemptTotals) {
|
|
239
|
+
const count = attemptCounts.get(idx) ?? 1
|
|
240
|
+
const avg = total / count
|
|
241
|
+
if (avg > bestAvg) {
|
|
242
|
+
bestAvg = avg
|
|
243
|
+
winnerIndex = idx
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// Phase 2: optional synthesis
|
|
248
|
+
let synthesis: string | undefined
|
|
249
|
+
if (opts.synthesize !== false) {
|
|
250
|
+
const winner = attempts[winnerIndex]
|
|
251
|
+
const runnerUps = attempts
|
|
252
|
+
.map((a, i) => ({ attempt: a, index: i }))
|
|
253
|
+
.filter((e) => e.index !== winnerIndex)
|
|
254
|
+
|
|
255
|
+
const synthPrompt =
|
|
256
|
+
`Synthesize the final result from the WINNING approach, grafting the best ideas from runner-ups.\n\n` +
|
|
257
|
+
`WINNER:\n${JSON.stringify(winner, null, 2)}\n\n` +
|
|
258
|
+
`RUNNER-UPS:\n${JSON.stringify(runnerUps, null, 2)}\n\n` +
|
|
259
|
+
`Provide a comprehensive synthesis combining the winner's structure with the best elements from other approaches.`
|
|
260
|
+
|
|
261
|
+
const synthResult = await agentFn(synthPrompt)
|
|
262
|
+
if (typeof synthResult === 'string') {
|
|
263
|
+
synthesis = synthResult
|
|
264
|
+
} else if (synthResult && typeof synthResult === 'object' && 'synthesis' in synthResult) {
|
|
265
|
+
synthesis = String((synthResult as Record<string, unknown>).synthesis)
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
return {
|
|
270
|
+
winner: attempts[winnerIndex],
|
|
271
|
+
winnerIndex,
|
|
272
|
+
scores,
|
|
273
|
+
synthesis,
|
|
274
|
+
}
|
|
275
|
+
}
|
package/src/workflow/runtime.ts
CHANGED
|
@@ -5,6 +5,8 @@ import { workflowAgent } from './primitives/agent'
|
|
|
5
5
|
import { parallel } from './primitives/parallel'
|
|
6
6
|
import { pipeline } from './primitives/pipeline'
|
|
7
7
|
import { phase as phasePrimitive } from './primitives/phase'
|
|
8
|
+
import { verify, judge } from './primitives/verify'
|
|
9
|
+
import { loopUntilConvergence } from './primitives/loop'
|
|
8
10
|
import type { ProviderRegistry } from '../providers/registry'
|
|
9
11
|
import type { QueryEngine } from '../core/engine'
|
|
10
12
|
|
|
@@ -147,6 +149,9 @@ export async function runWorkflow(
|
|
|
147
149
|
'agent',
|
|
148
150
|
'parallel',
|
|
149
151
|
'pipeline',
|
|
152
|
+
'verify',
|
|
153
|
+
'judge',
|
|
154
|
+
'loopUntilConvergence',
|
|
150
155
|
'phase',
|
|
151
156
|
'log',
|
|
152
157
|
'args',
|
|
@@ -154,7 +159,18 @@ export async function runWorkflow(
|
|
|
154
159
|
wrappedScript,
|
|
155
160
|
)
|
|
156
161
|
|
|
157
|
-
const result = await scriptFn(
|
|
162
|
+
const result = await scriptFn(
|
|
163
|
+
agent,
|
|
164
|
+
parallel,
|
|
165
|
+
pipeline,
|
|
166
|
+
verify,
|
|
167
|
+
judge,
|
|
168
|
+
loopUntilConvergence,
|
|
169
|
+
wrappedPhase,
|
|
170
|
+
log,
|
|
171
|
+
args,
|
|
172
|
+
budget,
|
|
173
|
+
)
|
|
158
174
|
|
|
159
175
|
// Count journal entries from state
|
|
160
176
|
const priorEntries = loadJournal(runId)
|