@miphamai/cli 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,103 @@
1
+ import { parallel } from './parallel'
2
+
3
+ /**
4
+ * loopUntilConvergence() — workflow primitive for iterative discovery with
5
+ * deduplication, adversarial verification, and convergence detection.
6
+ *
7
+ * Orchestrates multiple "finders" over successive rounds until no new items
8
+ * are discovered for `dryRounds` consecutive rounds (convergence), or
9
+ * `maxRounds` is reached (cutoff).
10
+ *
11
+ * Each item is keyed by `keyFn` for deduplication across rounds. If a
12
+ * `verify` function is provided, only items that survive verification
13
+ * enter the confirmed set — but all seen items (even those that fail
14
+ * verification) are tracked in the seen-set to prevent re-discovery.
15
+ */
16
+
17
+ export interface VerifyVote {
18
+ real: boolean
19
+ reason: string
20
+ }
21
+
22
+ export interface VerifyResult<T = unknown> {
23
+ finding: T
24
+ survives: boolean
25
+ votes: VerifyVote[]
26
+ score: number
27
+ }
28
+
29
+ export interface LoopUntilConvergenceResult<T> {
30
+ confirmed: T[]
31
+ totalSeen: number
32
+ rounds: number
33
+ converged: boolean
34
+ }
35
+
36
+ export interface LoopUntilConvergenceOpts<T> {
37
+ finders: Array<() => Promise<{ items: T[] } | null>>
38
+ keyFn: (item: T) => string
39
+ verify?: (item: T) => Promise<VerifyResult<T>>
40
+ dryRounds?: number
41
+ maxRounds?: number
42
+ }
43
+
44
+ export async function loopUntilConvergence<T>(
45
+ opts: LoopUntilConvergenceOpts<T>,
46
+ ): Promise<LoopUntilConvergenceResult<T>> {
47
+ const dryRounds = opts.dryRounds ?? 2
48
+ const maxRounds = opts.maxRounds ?? 20
49
+
50
+ const seen = new Set<string>()
51
+ const confirmed: T[] = []
52
+ let dry = 0
53
+ let rounds = 0
54
+
55
+ while (dry < dryRounds && rounds < maxRounds) {
56
+ rounds++
57
+
58
+ // FAN OUT: all finders run in parallel
59
+ const raw = await parallel(opts.finders.map((f) => () => f()))
60
+
61
+ // EDGE LOGIC: flatMap + dedup (pure JS, zero tokens)
62
+ const items: T[] = []
63
+ for (const result of raw) {
64
+ if (result && result.items) {
65
+ items.push(...result.items)
66
+ }
67
+ }
68
+
69
+ // Dedup against SEEN set, not confirmed
70
+ const fresh = items.filter((item) => {
71
+ const key = opts.keyFn(item)
72
+ if (seen.has(key)) return false
73
+ seen.add(key)
74
+ return true
75
+ })
76
+
77
+ if (fresh.length === 0) {
78
+ dry++ // no new unique items → trending toward convergence
79
+ continue
80
+ }
81
+
82
+ dry = 0 // new items found → reset dry counter
83
+
84
+ // VERIFY: optional quality gate
85
+ if (opts.verify) {
86
+ const judged = await parallel(fresh.map((item) => () => opts.verify!(item)))
87
+ for (const j of judged) {
88
+ if (j && j.survives) {
89
+ confirmed.push(j.finding as T)
90
+ }
91
+ }
92
+ } else {
93
+ confirmed.push(...fresh)
94
+ }
95
+ }
96
+
97
+ return {
98
+ confirmed,
99
+ totalSeen: seen.size,
100
+ rounds,
101
+ converged: dry >= dryRounds,
102
+ }
103
+ }
@@ -0,0 +1,275 @@
1
+ /**
2
+ * verify() and judge() — workflow primitives for adversarial verification,
3
+ * multi-perspective review, and multi-judge evaluation.
4
+ *
5
+ * verify() supports 3 modes:
6
+ * - adversarial: N skeptics try to refute the finding (majority wins)
7
+ * - perspective: N lenses each evaluate from a specific angle (at least 1 confirms)
8
+ * - consensus: N voters must unanimously agree
9
+ *
10
+ * judge() evaluates N attempts by M judges, computes average scores,
11
+ * picks the winner, and optionally synthesizes a final result.
12
+ */
13
+
14
+ import { workflowAgent } from './agent'
15
+ import { parallel } from './parallel'
16
+ import type { WorkflowAgentOpts } from './agent'
17
+
18
+ // Agent function signature matching the sandbox-injected pattern:
19
+ // (prompt, opts?) → result. In production, the workflow runtime binds
20
+ // ProviderRegistry + ToolRegistry into a function of this shape and
21
+ // injects it via _mockAgent.
22
+ type AgentFn = (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
23
+
24
+ // ── Types ──
25
+
26
+ export interface VerifyResult {
27
+ finding: unknown
28
+ survives: boolean
29
+ votes: Array<{ real: boolean; reason: string; lens?: string }>
30
+ score: number
31
+ }
32
+
33
+ export type VerifyMode = 'adversarial' | 'perspective' | 'consensus'
34
+
35
+ export interface VerifyOpts {
36
+ mode: VerifyMode
37
+ skeptics?: number // adversarial: default 3
38
+ lenses?: string[] // perspective: e.g. ['correctness', 'security']
39
+ voters?: number // consensus: default 3
40
+ threshold?: number
41
+ schema: Record<string, unknown>
42
+ /** Test-only: inject a mock agent function. When not provided, falls back to workflowAgent. */
43
+ _mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
44
+ }
45
+
46
+ interface VerdictVote {
47
+ real: boolean
48
+ reason: string
49
+ lens?: string
50
+ }
51
+
52
+ export interface JudgeResult {
53
+ winner: unknown
54
+ winnerIndex: number
55
+ scores: Array<{
56
+ attemptIndex: number
57
+ judgeIndex: number
58
+ criteria: Record<string, number>
59
+ total: number
60
+ notes: string
61
+ }>
62
+ synthesis?: string
63
+ }
64
+
65
+ export interface JudgeOpts {
66
+ criteria: string[]
67
+ judges?: number // default: 3
68
+ synthesize?: boolean // default: true
69
+ schema: Record<string, unknown>
70
+ /** Test-only: inject a mock agent function. */
71
+ _mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
72
+ }
73
+
74
+ // ── Helpers ──
75
+
76
+ function defaultThreshold(mode: VerifyMode, total: number): number {
77
+ if (mode === 'consensus') return total // all must agree
78
+ if (mode === 'perspective') return 1 // at least one lens confirms
79
+ return Math.ceil(total / 2) // adversarial: majority
80
+ }
81
+
82
+ // ── verify() ──
83
+
84
+ export async function verify(finding: unknown, opts: VerifyOpts): Promise<VerifyResult> {
85
+ // Prefer the test hook (_mockAgent), fall back to workflowAgent.
86
+ // In production sandbox usage, _mockAgent is set to the pre-bound agent
87
+ // function injected by the workflow runtime.
88
+ const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
89
+ const mode = opts.mode
90
+
91
+ const findingStr = JSON.stringify(finding, null, 2)
92
+ const schemaDesc = JSON.stringify(opts.schema)
93
+
94
+ let prompts: Array<{ prompt: string; lens?: string }>
95
+
96
+ switch (mode) {
97
+ case 'adversarial': {
98
+ const count = opts.skeptics ?? 3
99
+ prompts = Array.from({ length: count }, (_, i) => ({
100
+ prompt:
101
+ `You are a skeptical reviewer (skeptic #${i + 1}). Try to REFUTE this finding. Default to real=false if uncertain.\n\n` +
102
+ `Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
103
+ }))
104
+ break
105
+ }
106
+ case 'perspective': {
107
+ const lenses = opts.lenses ?? ['correctness']
108
+ prompts = lenses.map((lens) => ({
109
+ prompt:
110
+ `Judge this finding through the "${lens}" lens. Is it valid from this perspective?\n\n` +
111
+ `Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
112
+ lens,
113
+ }))
114
+ break
115
+ }
116
+ case 'consensus': {
117
+ const count = opts.voters ?? 3
118
+ prompts = Array.from({ length: count }, () => ({
119
+ prompt:
120
+ `Is this finding correct? Be honest and critical. Vote real=true only if you are fully convinced.\n\n` +
121
+ `Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
122
+ }))
123
+ break
124
+ }
125
+ }
126
+
127
+ // Fan out all agent calls concurrently
128
+ const rawVotes = await parallel(
129
+ prompts.map(
130
+ (p) => () => agentFn(p.prompt, { schema: opts.schema as WorkflowAgentOpts['schema'] }),
131
+ ),
132
+ )
133
+
134
+ // Filter out failed agents (null) and build vote objects
135
+ const votes: VerdictVote[] = []
136
+ for (let i = 0; i < rawVotes.length; i++) {
137
+ const v = rawVotes[i]
138
+ if (v && typeof v === 'object') {
139
+ const vote = v as Record<string, unknown>
140
+ votes.push({
141
+ real: Boolean(vote.real),
142
+ reason: String(vote.reason ?? ''),
143
+ lens: prompts[i]?.lens,
144
+ })
145
+ }
146
+ }
147
+
148
+ const threshold = opts.threshold ?? defaultThreshold(mode, votes.length)
149
+ const realCount = votes.filter((v) => v.real).length
150
+ const survives = realCount >= threshold
151
+
152
+ return {
153
+ finding,
154
+ survives,
155
+ votes,
156
+ score: votes.length > 0 ? realCount / votes.length : 0,
157
+ }
158
+ }
159
+
160
+ // ── judge() ──
161
+
162
+ export async function judge(attempts: unknown[], opts: JudgeOpts): Promise<JudgeResult> {
163
+ // Prefer the test hook, fall back to workflowAgent.
164
+ const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
165
+ const judgeCount = opts.judges ?? 3
166
+ const schemaDesc = JSON.stringify(opts.schema)
167
+
168
+ // Phase 1: each judge scores each attempt
169
+ interface ScoreEntry {
170
+ attemptIndex: number
171
+ judgeIndex: number
172
+ criteria: Record<string, number>
173
+ total: number
174
+ notes: string
175
+ }
176
+
177
+ // Build the cross-product of judges × attempts
178
+ const scorePrompts: Array<{
179
+ attempt: unknown
180
+ attemptIndex: number
181
+ judgeIndex: number
182
+ }> = []
183
+ for (let ji = 0; ji < judgeCount; ji++) {
184
+ for (let ai = 0; ai < attempts.length; ai++) {
185
+ scorePrompts.push({
186
+ attempt: attempts[ai],
187
+ attemptIndex: ai,
188
+ judgeIndex: ji,
189
+ })
190
+ }
191
+ }
192
+
193
+ // Fan out all scoring calls concurrently
194
+ const rawScores = await parallel(
195
+ scorePrompts.map(
196
+ (sp) => () =>
197
+ agentFn(
198
+ `You are judge #${sp.judgeIndex + 1}. Score this attempt against the criteria: ${opts.criteria.join(', ')}.\n\n` +
199
+ `Attempt:\n${JSON.stringify(sp.attempt, null, 2)}\n\n` +
200
+ `Return JSON matching this schema:\n${schemaDesc}`,
201
+ { schema: opts.schema as WorkflowAgentOpts['schema'] },
202
+ ),
203
+ ),
204
+ )
205
+
206
+ // Parse and validate each score result
207
+ const scores: ScoreEntry[] = []
208
+ for (let i = 0; i < rawScores.length; i++) {
209
+ const raw = rawScores[i]
210
+ const sp = scorePrompts[i]!
211
+ if (raw && typeof raw === 'object') {
212
+ const obj = raw as Record<string, unknown>
213
+ const criteriaObj = (obj.scores as Record<string, number>) ?? {}
214
+ const total = Object.values(criteriaObj).reduce(
215
+ (sum, v) => sum + (typeof v === 'number' ? v : 0),
216
+ 0,
217
+ )
218
+ scores.push({
219
+ attemptIndex: sp.attemptIndex,
220
+ judgeIndex: sp.judgeIndex,
221
+ criteria: criteriaObj,
222
+ total,
223
+ notes: String(obj.notes ?? ''),
224
+ })
225
+ }
226
+ }
227
+
228
+ // Compute winner: highest average total across judges
229
+ const attemptTotals = new Map<number, number>()
230
+ const attemptCounts = new Map<number, number>()
231
+ for (const s of scores) {
232
+ attemptTotals.set(s.attemptIndex, (attemptTotals.get(s.attemptIndex) ?? 0) + s.total)
233
+ attemptCounts.set(s.attemptIndex, (attemptCounts.get(s.attemptIndex) ?? 0) + 1)
234
+ }
235
+
236
+ let winnerIndex = 0
237
+ let bestAvg = -Infinity
238
+ for (const [idx, total] of attemptTotals) {
239
+ const count = attemptCounts.get(idx) ?? 1
240
+ const avg = total / count
241
+ if (avg > bestAvg) {
242
+ bestAvg = avg
243
+ winnerIndex = idx
244
+ }
245
+ }
246
+
247
+ // Phase 2: optional synthesis
248
+ let synthesis: string | undefined
249
+ if (opts.synthesize !== false) {
250
+ const winner = attempts[winnerIndex]
251
+ const runnerUps = attempts
252
+ .map((a, i) => ({ attempt: a, index: i }))
253
+ .filter((e) => e.index !== winnerIndex)
254
+
255
+ const synthPrompt =
256
+ `Synthesize the final result from the WINNING approach, grafting the best ideas from runner-ups.\n\n` +
257
+ `WINNER:\n${JSON.stringify(winner, null, 2)}\n\n` +
258
+ `RUNNER-UPS:\n${JSON.stringify(runnerUps, null, 2)}\n\n` +
259
+ `Provide a comprehensive synthesis combining the winner's structure with the best elements from other approaches.`
260
+
261
+ const synthResult = await agentFn(synthPrompt)
262
+ if (typeof synthResult === 'string') {
263
+ synthesis = synthResult
264
+ } else if (synthResult && typeof synthResult === 'object' && 'synthesis' in synthResult) {
265
+ synthesis = String((synthResult as Record<string, unknown>).synthesis)
266
+ }
267
+ }
268
+
269
+ return {
270
+ winner: attempts[winnerIndex],
271
+ winnerIndex,
272
+ scores,
273
+ synthesis,
274
+ }
275
+ }
@@ -5,6 +5,8 @@ import { workflowAgent } from './primitives/agent'
5
5
  import { parallel } from './primitives/parallel'
6
6
  import { pipeline } from './primitives/pipeline'
7
7
  import { phase as phasePrimitive } from './primitives/phase'
8
+ import { verify, judge } from './primitives/verify'
9
+ import { loopUntilConvergence } from './primitives/loop'
8
10
  import type { ProviderRegistry } from '../providers/registry'
9
11
  import type { QueryEngine } from '../core/engine'
10
12
 
@@ -147,6 +149,9 @@ export async function runWorkflow(
147
149
  'agent',
148
150
  'parallel',
149
151
  'pipeline',
152
+ 'verify',
153
+ 'judge',
154
+ 'loopUntilConvergence',
150
155
  'phase',
151
156
  'log',
152
157
  'args',
@@ -154,7 +159,18 @@ export async function runWorkflow(
154
159
  wrappedScript,
155
160
  )
156
161
 
157
- const result = await scriptFn(agent, parallel, pipeline, wrappedPhase, log, args, budget)
162
+ const result = await scriptFn(
163
+ agent,
164
+ parallel,
165
+ pipeline,
166
+ verify,
167
+ judge,
168
+ loopUntilConvergence,
169
+ wrappedPhase,
170
+ log,
171
+ args,
172
+ budget,
173
+ )
158
174
 
159
175
  // Count journal entries from state
160
176
  const priorEntries = loadJournal(runId)