@miphamai/cli 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/workflows/audit.js +51 -0
- package/skills/workflows/hunt.js +95 -0
- package/skills/workflows/judge.js +55 -0
- package/skills/workflows/migrate.js +92 -0
- package/skills/workflows/research.js +87 -0
- package/skills/workflows/review.js +92 -0
- package/src/core/instructions.ts +28 -0
- package/src/tools/agent/workflow.ts +67 -3
- package/src/ui/commands.ts +106 -0
- package/src/workflow/primitives/loop.ts +103 -0
- package/src/workflow/primitives/verify.ts +275 -0
- package/src/workflow/runtime.ts +17 -1
package/package.json
CHANGED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'audit',
|
|
3
|
+
description: 'Security audit: fan-out per file → verify → report',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Scope', detail: 'discover targets' },
|
|
6
|
+
{ title: 'Audit', detail: 'one agent per target' },
|
|
7
|
+
{ title: 'Verify', detail: 'adversarial verification' },
|
|
8
|
+
{ title: 'Report', detail: 'synthesize findings' },
|
|
9
|
+
],
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
phase('Scope')
|
|
13
|
+
const targets = args.targets || (await agent(
|
|
14
|
+
'List all source files that need security auditing. Return { files: [{ path, reason }] }',
|
|
15
|
+
{ schema: { type: 'object', properties: { files: { type: 'array', items: { type: 'object', properties: { path: { type: 'string' }, reason: { type: 'string' } }, required: ['path', 'reason'] } } }, required: ['files'] } },
|
|
16
|
+
)).files
|
|
17
|
+
|
|
18
|
+
log(`Auditing ${targets.length} files`)
|
|
19
|
+
|
|
20
|
+
phase('Audit')
|
|
21
|
+
const raw = (await pipeline(
|
|
22
|
+
targets,
|
|
23
|
+
t => agent(`Security audit ${t.path}: injection, auth, crypto, secrets, input validation. Return { findings: [{ severity, file, line, summary }] }`,
|
|
24
|
+
{ label: `audit:${t.path}`, schema: { type: 'object', properties: { findings: { type: 'array', items: { type: 'object', properties: { severity: { type: 'string', enum: ['critical', 'high', 'medium', 'low'] }, file: { type: 'string' }, line: { type: 'number' }, summary: { type: 'string' } }, required: ['severity', 'file', 'summary'] } } }, required: ['findings'] } },
|
|
25
|
+
)),
|
|
26
|
+
)
|
|
27
|
+
const findings = raw.flatMap(r => (r && r.findings) || [])
|
|
28
|
+
|
|
29
|
+
log(`Found ${findings.length} potential issues`)
|
|
30
|
+
|
|
31
|
+
phase('Verify')
|
|
32
|
+
const verified = await parallel(
|
|
33
|
+
findings.map(f => () =>
|
|
34
|
+
verify(f, {
|
|
35
|
+
mode: 'adversarial',
|
|
36
|
+
skeptics: 3,
|
|
37
|
+
threshold: 2,
|
|
38
|
+
schema: { type: 'object', properties: { real: { type: 'boolean' }, reason: { type: 'string' } }, required: ['real', 'reason'] },
|
|
39
|
+
})
|
|
40
|
+
),
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
const confirmed = verified.filter(Boolean).filter(v => v.survives).map(v => v.finding)
|
|
44
|
+
log(`${confirmed.length}/${findings.length} findings verified`)
|
|
45
|
+
|
|
46
|
+
phase('Report')
|
|
47
|
+
const report = await agent(
|
|
48
|
+
`Synthesize audit report from confirmed findings:\n${JSON.stringify(confirmed)}`,
|
|
49
|
+
{ label: 'report' },
|
|
50
|
+
)
|
|
51
|
+
return report
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'hunt',
|
|
3
|
+
description: 'Bug hunt: loopUntilConvergence + adversarial verify',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Hunt', detail: 'iterative discovery + verify' },
|
|
6
|
+
{ title: 'Report', detail: 'synthesize findings' },
|
|
7
|
+
],
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
phase('Hunt')
|
|
11
|
+
const target = args.target || 'this codebase'
|
|
12
|
+
|
|
13
|
+
const { confirmed, totalSeen, rounds, converged } = await loopUntilConvergence({
|
|
14
|
+
finders: [
|
|
15
|
+
() =>
|
|
16
|
+
agent(
|
|
17
|
+
`Find bugs in ${target}. Look for: null safety, race conditions, resource leaks, edge cases. Return { items: [{ file, line, summary, type }] }`,
|
|
18
|
+
{
|
|
19
|
+
label: 'hunt:general',
|
|
20
|
+
schema: {
|
|
21
|
+
type: 'object',
|
|
22
|
+
properties: {
|
|
23
|
+
items: {
|
|
24
|
+
type: 'array',
|
|
25
|
+
items: {
|
|
26
|
+
type: 'object',
|
|
27
|
+
properties: {
|
|
28
|
+
file: { type: 'string' },
|
|
29
|
+
line: { type: 'number' },
|
|
30
|
+
summary: { type: 'string' },
|
|
31
|
+
type: { type: 'string' },
|
|
32
|
+
},
|
|
33
|
+
required: ['file', 'summary', 'type'],
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
required: ['items'],
|
|
38
|
+
},
|
|
39
|
+
},
|
|
40
|
+
),
|
|
41
|
+
() =>
|
|
42
|
+
agent(
|
|
43
|
+
`Find security vulnerabilities in ${target}: injection, auth bypass, insecure crypto, exposed secrets. Return { items: [{ file, line, summary, type }] }`,
|
|
44
|
+
{
|
|
45
|
+
label: 'hunt:security',
|
|
46
|
+
schema: {
|
|
47
|
+
type: 'object',
|
|
48
|
+
properties: {
|
|
49
|
+
items: {
|
|
50
|
+
type: 'array',
|
|
51
|
+
items: {
|
|
52
|
+
type: 'object',
|
|
53
|
+
properties: {
|
|
54
|
+
file: { type: 'string' },
|
|
55
|
+
line: { type: 'number' },
|
|
56
|
+
summary: { type: 'string' },
|
|
57
|
+
type: { type: 'string' },
|
|
58
|
+
},
|
|
59
|
+
required: ['file', 'summary', 'type'],
|
|
60
|
+
},
|
|
61
|
+
},
|
|
62
|
+
},
|
|
63
|
+
required: ['items'],
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
),
|
|
67
|
+
],
|
|
68
|
+
keyFn: (bug) => `${bug.file}:${bug.line}:${bug.summary}`,
|
|
69
|
+
verify: async (bug) =>
|
|
70
|
+
verify(bug, {
|
|
71
|
+
mode: 'adversarial',
|
|
72
|
+
skeptics: 3,
|
|
73
|
+
threshold: 2,
|
|
74
|
+
schema: {
|
|
75
|
+
type: 'object',
|
|
76
|
+
properties: { real: { type: 'boolean' }, reason: { type: 'string' } },
|
|
77
|
+
required: ['real', 'reason'],
|
|
78
|
+
},
|
|
79
|
+
}),
|
|
80
|
+
dryRounds: 2,
|
|
81
|
+
maxRounds: 10,
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
log(
|
|
85
|
+
`${converged ? 'Converged' : 'Max rounds reached'} after ${rounds} rounds. ${totalSeen} unique bugs seen, ${confirmed.length} confirmed.`,
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
phase('Report')
|
|
89
|
+
if (confirmed.length === 0) {
|
|
90
|
+
return `No confirmed bugs found after ${rounds} rounds of hunting.`
|
|
91
|
+
}
|
|
92
|
+
return await agent(
|
|
93
|
+
`Write a bug report from ${confirmed.length} confirmed bugs:\n${JSON.stringify(confirmed)}`,
|
|
94
|
+
{ label: 'report' },
|
|
95
|
+
)
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'judge',
|
|
3
|
+
description: 'Judge panel: N competing approaches × M judges → winner + synthesis',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Generate', detail: 'generate competing approaches' },
|
|
6
|
+
{ title: 'Judge', detail: 'score and rank' },
|
|
7
|
+
{ title: 'Synthesize', detail: 'final recommendation' },
|
|
8
|
+
],
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
phase('Generate')
|
|
12
|
+
const problem = args.problem || (await agent('What problem are we solving?', { label: 'query' }))
|
|
13
|
+
|
|
14
|
+
const approaches = await parallel([
|
|
15
|
+
() => agent(`Solve "${problem}" with an MVP-first approach.`, { label: 'gen:mvp' }),
|
|
16
|
+
() => agent(`Solve "${problem}" with a risk-first approach.`, { label: 'gen:risk' }),
|
|
17
|
+
() => agent(`Solve "${problem}" with a user-first approach.`, { label: 'gen:user' }),
|
|
18
|
+
])
|
|
19
|
+
|
|
20
|
+
const validApproaches = approaches.filter(Boolean)
|
|
21
|
+
log(`Generated ${validApproaches.length} approaches`)
|
|
22
|
+
|
|
23
|
+
phase('Judge')
|
|
24
|
+
const { winner, winnerIndex, scores, synthesis } = await judge(validApproaches, {
|
|
25
|
+
criteria: ['feasibility', 'impact', 'simplicity', 'risk'],
|
|
26
|
+
judges: 3,
|
|
27
|
+
synthesize: true,
|
|
28
|
+
schema: {
|
|
29
|
+
type: 'object',
|
|
30
|
+
properties: {
|
|
31
|
+
scores: {
|
|
32
|
+
type: 'object',
|
|
33
|
+
properties: {
|
|
34
|
+
feasibility: { type: 'number' },
|
|
35
|
+
impact: { type: 'number' },
|
|
36
|
+
simplicity: { type: 'number' },
|
|
37
|
+
risk: { type: 'number' },
|
|
38
|
+
},
|
|
39
|
+
required: ['feasibility', 'impact', 'simplicity', 'risk'],
|
|
40
|
+
},
|
|
41
|
+
notes: { type: 'string' },
|
|
42
|
+
},
|
|
43
|
+
required: ['scores', 'notes'],
|
|
44
|
+
},
|
|
45
|
+
})
|
|
46
|
+
|
|
47
|
+
phase('Synthesize')
|
|
48
|
+
return {
|
|
49
|
+
problem,
|
|
50
|
+
winner: `Approach #${winnerIndex + 1}`,
|
|
51
|
+
scores_summary: scores.map(
|
|
52
|
+
(s) => `Judge ${s.judgeIndex + 1}: approach ${s.attemptIndex + 1} = ${s.total}`,
|
|
53
|
+
),
|
|
54
|
+
synthesis,
|
|
55
|
+
}
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'migrate',
|
|
3
|
+
description: 'Code migration: discover → fan-out transform → verify → integrate',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Discover', detail: 'find migration targets' },
|
|
6
|
+
{ title: 'Transform', detail: 'one agent per file' },
|
|
7
|
+
{ title: 'Verify', detail: 'validate transformations' },
|
|
8
|
+
],
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
phase('Discover')
|
|
12
|
+
const pattern =
|
|
13
|
+
args.pattern ||
|
|
14
|
+
(await agent('What code pattern needs migration? Return { pattern, replacement, reason }', {
|
|
15
|
+
schema: {
|
|
16
|
+
type: 'object',
|
|
17
|
+
properties: {
|
|
18
|
+
pattern: { type: 'string' },
|
|
19
|
+
replacement: { type: 'string' },
|
|
20
|
+
reason: { type: 'string' },
|
|
21
|
+
},
|
|
22
|
+
required: ['pattern', 'replacement'],
|
|
23
|
+
},
|
|
24
|
+
}))
|
|
25
|
+
|
|
26
|
+
const files = await agent(
|
|
27
|
+
`Find all files matching pattern: ${pattern.pattern}. Return { files: [{ path }] }`,
|
|
28
|
+
{
|
|
29
|
+
schema: {
|
|
30
|
+
type: 'object',
|
|
31
|
+
properties: {
|
|
32
|
+
files: {
|
|
33
|
+
type: 'array',
|
|
34
|
+
items: { type: 'object', properties: { path: { type: 'string' } }, required: ['path'] },
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
required: ['files'],
|
|
38
|
+
},
|
|
39
|
+
},
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
log(`Migrating ${files.files.length} files: ${pattern.pattern} → ${pattern.replacement}`)
|
|
43
|
+
|
|
44
|
+
phase('Transform')
|
|
45
|
+
const results = await pipeline(files.files, (f) =>
|
|
46
|
+
agent(
|
|
47
|
+
`In ${f.path}, migrate "${pattern.pattern}" to "${pattern.replacement}". Reason: ${pattern.reason}. Return { path, changes, success }`,
|
|
48
|
+
{
|
|
49
|
+
label: `migrate:${f.path}`,
|
|
50
|
+
isolation: 'worktree',
|
|
51
|
+
schema: {
|
|
52
|
+
type: 'object',
|
|
53
|
+
properties: {
|
|
54
|
+
path: { type: 'string' },
|
|
55
|
+
changes: { type: 'number' },
|
|
56
|
+
success: { type: 'boolean' },
|
|
57
|
+
},
|
|
58
|
+
required: ['path', 'success'],
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
const succeeded = results.filter(Boolean).filter((r) => r.success)
|
|
65
|
+
const failed = results.filter(Boolean).filter((r) => !r.success)
|
|
66
|
+
log(`${succeeded.length} migrated, ${failed.length} failed`)
|
|
67
|
+
|
|
68
|
+
phase('Verify')
|
|
69
|
+
const verified = await parallel(
|
|
70
|
+
succeeded.map(
|
|
71
|
+
(f) => () =>
|
|
72
|
+
verify(
|
|
73
|
+
{ file: f.path, migration: pattern.replacement },
|
|
74
|
+
{
|
|
75
|
+
mode: 'perspective',
|
|
76
|
+
lenses: ['correctness', 'style'],
|
|
77
|
+
threshold: 1,
|
|
78
|
+
schema: {
|
|
79
|
+
type: 'object',
|
|
80
|
+
properties: { real: { type: 'boolean' }, reason: { type: 'string' } },
|
|
81
|
+
required: ['real', 'reason'],
|
|
82
|
+
},
|
|
83
|
+
},
|
|
84
|
+
),
|
|
85
|
+
),
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
return {
|
|
89
|
+
migrated: succeeded.length,
|
|
90
|
+
failed: failed.length,
|
|
91
|
+
verified: verified.filter(Boolean).filter((v) => v.survives).length,
|
|
92
|
+
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'research',
|
|
3
|
+
description: 'Deep research: scope → parallel search → verify → synthesize',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Scope', detail: 'define research angles' },
|
|
6
|
+
{ title: 'Research', detail: 'parallel web searches' },
|
|
7
|
+
{ title: 'Verify', detail: 'adversarial verification' },
|
|
8
|
+
{ title: 'Synthesize', detail: 'final report' },
|
|
9
|
+
],
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
phase('Scope')
|
|
13
|
+
const topic = args.topic || (await agent('What topic should we research?', { label: 'query' }))
|
|
14
|
+
|
|
15
|
+
phase('Research')
|
|
16
|
+
const angles = ['overview', 'technical-details', 'competitors', 'criticism', 'future-trends']
|
|
17
|
+
const raw = await parallel(
|
|
18
|
+
angles.map(
|
|
19
|
+
(angle) => () =>
|
|
20
|
+
agent(
|
|
21
|
+
`Research "${topic}" from angle: ${angle}. Return { sources: [{ title, url, keyPoint }] }`,
|
|
22
|
+
{
|
|
23
|
+
label: `research:${angle}`,
|
|
24
|
+
schema: {
|
|
25
|
+
type: 'object',
|
|
26
|
+
properties: {
|
|
27
|
+
sources: {
|
|
28
|
+
type: 'array',
|
|
29
|
+
items: {
|
|
30
|
+
type: 'object',
|
|
31
|
+
properties: {
|
|
32
|
+
title: { type: 'string' },
|
|
33
|
+
url: { type: 'string' },
|
|
34
|
+
keyPoint: { type: 'string' },
|
|
35
|
+
},
|
|
36
|
+
required: ['title', 'keyPoint'],
|
|
37
|
+
},
|
|
38
|
+
},
|
|
39
|
+
},
|
|
40
|
+
required: ['sources'],
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
),
|
|
44
|
+
),
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
const allSources = raw.filter(Boolean).flatMap((r) => r.sources)
|
|
48
|
+
const seen = new Set()
|
|
49
|
+
const unique = allSources.filter((s) => {
|
|
50
|
+
const k = s.url
|
|
51
|
+
if (seen.has(k)) return false
|
|
52
|
+
seen.add(k)
|
|
53
|
+
return true
|
|
54
|
+
})
|
|
55
|
+
log(`Collected ${unique.length} unique sources`)
|
|
56
|
+
|
|
57
|
+
phase('Verify')
|
|
58
|
+
const verified = await parallel(
|
|
59
|
+
unique.map(
|
|
60
|
+
(s) => () =>
|
|
61
|
+
verify(
|
|
62
|
+
{ claim: s.keyPoint, source: s.title },
|
|
63
|
+
{
|
|
64
|
+
mode: 'adversarial',
|
|
65
|
+
skeptics: 2,
|
|
66
|
+
threshold: 1,
|
|
67
|
+
schema: {
|
|
68
|
+
type: 'object',
|
|
69
|
+
properties: { real: { type: 'boolean' }, reason: { type: 'string' } },
|
|
70
|
+
required: ['real', 'reason'],
|
|
71
|
+
},
|
|
72
|
+
},
|
|
73
|
+
),
|
|
74
|
+
),
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
const credible = verified
|
|
78
|
+
.filter(Boolean)
|
|
79
|
+
.filter((v) => v.survives)
|
|
80
|
+
.map((v) => v.finding)
|
|
81
|
+
|
|
82
|
+
phase('Synthesize')
|
|
83
|
+
const report = await agent(
|
|
84
|
+
`Synthesize research report on "${topic}" from credible sources:\n${JSON.stringify(credible)}`,
|
|
85
|
+
{ label: 'synthesize' },
|
|
86
|
+
)
|
|
87
|
+
return report
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'review',
|
|
3
|
+
description: 'Code review: fan-out per dimension → judge panel → report',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Review', detail: 'multi-dimensional review' },
|
|
6
|
+
{ title: 'Judge', detail: 'judge panel scores findings' },
|
|
7
|
+
{ title: 'Report', detail: 'final review report' },
|
|
8
|
+
],
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
phase('Review')
|
|
12
|
+
const dimensions = ['correctness', 'security', 'performance', 'maintainability']
|
|
13
|
+
const rawFindings = await parallel(
|
|
14
|
+
dimensions.map(
|
|
15
|
+
(d) => () =>
|
|
16
|
+
agent(
|
|
17
|
+
`Review the code from the "${d}" lens. Return { findings: [{ severity, file, line, summary }] }`,
|
|
18
|
+
{
|
|
19
|
+
label: `review:${d}`,
|
|
20
|
+
schema: {
|
|
21
|
+
type: 'object',
|
|
22
|
+
properties: {
|
|
23
|
+
findings: {
|
|
24
|
+
type: 'array',
|
|
25
|
+
items: {
|
|
26
|
+
type: 'object',
|
|
27
|
+
properties: {
|
|
28
|
+
severity: { type: 'string', enum: ['blocker', 'high', 'medium', 'low'] },
|
|
29
|
+
file: { type: 'string' },
|
|
30
|
+
line: { type: 'number' },
|
|
31
|
+
summary: { type: 'string' },
|
|
32
|
+
dimension: { type: 'string' },
|
|
33
|
+
},
|
|
34
|
+
required: ['severity', 'file', 'summary'],
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
},
|
|
38
|
+
required: ['findings'],
|
|
39
|
+
},
|
|
40
|
+
},
|
|
41
|
+
),
|
|
42
|
+
),
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
// Edge logic: dedup across dimensions (pure JS)
|
|
46
|
+
const allFindings = rawFindings.filter(Boolean).flatMap((r) => r.findings)
|
|
47
|
+
const seen = new Set()
|
|
48
|
+
const uniqueFindings = allFindings.filter((f) => {
|
|
49
|
+
const k = `${f.file}:${f.line}:${f.summary}`
|
|
50
|
+
if (seen.has(k)) return false
|
|
51
|
+
seen.add(k)
|
|
52
|
+
return true
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
log(`${uniqueFindings.length} unique findings across ${dimensions.length} dimensions`)
|
|
56
|
+
|
|
57
|
+
phase('Judge')
|
|
58
|
+
if (uniqueFindings.length === 0) {
|
|
59
|
+
return 'No findings — code looks clean across all dimensions.'
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Judge panel: rank findings by severity
|
|
63
|
+
const judged = await judge(
|
|
64
|
+
uniqueFindings.map((f) => ({ finding: f })),
|
|
65
|
+
{
|
|
66
|
+
criteria: ['severity', 'actionability', 'confidence'],
|
|
67
|
+
judges: 2,
|
|
68
|
+
synthesize: false,
|
|
69
|
+
schema: {
|
|
70
|
+
type: 'object',
|
|
71
|
+
properties: {
|
|
72
|
+
scores: {
|
|
73
|
+
type: 'object',
|
|
74
|
+
properties: {
|
|
75
|
+
severity: { type: 'number' },
|
|
76
|
+
actionability: { type: 'number' },
|
|
77
|
+
confidence: { type: 'number' },
|
|
78
|
+
},
|
|
79
|
+
required: ['severity', 'actionability', 'confidence'],
|
|
80
|
+
},
|
|
81
|
+
notes: { type: 'string' },
|
|
82
|
+
},
|
|
83
|
+
required: ['scores', 'notes'],
|
|
84
|
+
},
|
|
85
|
+
},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
phase('Report')
|
|
89
|
+
return await agent(
|
|
90
|
+
`Write a code review report from ${uniqueFindings.length} findings (ranked by judge panel). Top finding: ${JSON.stringify(judged.winner)}`,
|
|
91
|
+
{ label: 'report' },
|
|
92
|
+
)
|
package/src/core/instructions.ts
CHANGED
|
@@ -66,6 +66,34 @@ export class InstructionsLoader {
|
|
|
66
66
|
parts.push(this.skillsReminder)
|
|
67
67
|
}
|
|
68
68
|
|
|
69
|
+
// Inject workflow auto-generation guidance
|
|
70
|
+
parts.push(`## Workflow Auto-Generation
|
|
71
|
+
|
|
72
|
+
When a task involves 3+ independent subtasks, multi-file operations,
|
|
73
|
+
or unknown-size discovery, generate a workflow script and execute it
|
|
74
|
+
via the Workflow tool instead of running agents sequentially. The
|
|
75
|
+
orchestration itself is code (zero tokens for inter-agent coordination).
|
|
76
|
+
|
|
77
|
+
Prefer workflows for: audits across many files, web research with multiple
|
|
78
|
+
sources, code migrations touching many files, security scans, bug hunts
|
|
79
|
+
with unknown scope, multi-dimensional code reviews.
|
|
80
|
+
|
|
81
|
+
Available primitives: agent(), parallel(), pipeline(), verify(),
|
|
82
|
+
judge(), loopUntilConvergence(), phase(), log(), args, budget.
|
|
83
|
+
|
|
84
|
+
Key rules:
|
|
85
|
+
- Default to pipeline() — only use parallel() barrier when a stage
|
|
86
|
+
genuinely needs all prior results at once
|
|
87
|
+
- Edge logic (flatten, dedupe, filter) is plain JS — not agent calls
|
|
88
|
+
- Use verify() on edges where confidence matters
|
|
89
|
+
- Use loopUntilConvergence() for discovery tasks with unknown size
|
|
90
|
+
|
|
91
|
+
When a workflow completes successfully, offer to save it:
|
|
92
|
+
"Workflow complete. Save this script? /workflow save <name>"
|
|
93
|
+
|
|
94
|
+
Script format: export const meta = { name, description, phases: [...] }
|
|
95
|
+
// script body using primitives...`)
|
|
96
|
+
|
|
69
97
|
return parts.join('\n\n---\n\n')
|
|
70
98
|
}
|
|
71
99
|
|
|
@@ -5,9 +5,56 @@ import type { QueryEngine } from '../../core/engine'
|
|
|
5
5
|
export const workflowTool: ToolDefinition = {
|
|
6
6
|
name: 'Workflow',
|
|
7
7
|
description:
|
|
8
|
-
'Execute a
|
|
9
|
-
'
|
|
10
|
-
'and
|
|
8
|
+
'Execute a workflow script that orchestrates multiple subagents deterministically. ' +
|
|
9
|
+
'Workflows run in the background — this tool returns immediately with a task ID, ' +
|
|
10
|
+
'and a <task-notification> arrives when the workflow completes. Use /workflows to watch live progress.\n\n' +
|
|
11
|
+
'A workflow structures work across many agents — to be comprehensive (decompose and cover in parallel), ' +
|
|
12
|
+
'to be confident (independent perspectives and adversarial checks before committing), ' +
|
|
13
|
+
'or to take on scale one context cannot hold (migrations, audits, broad sweeps). ' +
|
|
14
|
+
'The script is where you encode that structure: what fans out, what verifies, what synthesizes.\n\n' +
|
|
15
|
+
'ONLY call this tool when the task benefits from multi-agent orchestration. ' +
|
|
16
|
+
'For a simple single-agent lookup or edit, use the Agent tool or direct tools instead.\n\n' +
|
|
17
|
+
'## Primitives\n\n' +
|
|
18
|
+
'- agent(prompt: string, opts?: {label?, phase?, schema?, model?, effort?, isolation?}): Promise<any> — spawn a subagent. ' +
|
|
19
|
+
'Without schema, returns final text as string. With schema (JSON Schema), returns validated object — retries on mismatch.\n' +
|
|
20
|
+
'- parallel(thunks: Array<() => Promise<any>>): Promise<any[]> — BARRIER: runs all thunks concurrently, waits for all. ' +
|
|
21
|
+
'Failed thunks resolve to null. Use filter(Boolean) before consuming results.\n' +
|
|
22
|
+
'- pipeline(items: T[], ...stages): Promise<any[]> — NO barrier: each item flows through all stages independently. ' +
|
|
23
|
+
'Item A can be in stage 3 while item B is still in stage 1. DEFAULT choice for multi-stage work.\n' +
|
|
24
|
+
'- verify(finding, opts): Promise<VerifyResult> — adversarial/perspective/consensus quality gate. ' +
|
|
25
|
+
'Spawns skeptics or lens-based judges, applies threshold, returns {survives, votes, score}.\n' +
|
|
26
|
+
'- judge(attempts, opts): Promise<JudgeResult> — judge panel: N attempts scored by M judges across K criteria. ' +
|
|
27
|
+
'Returns winner with optional synthesis grafting runner-up ideas.\n' +
|
|
28
|
+
'- loopUntilConvergence(opts): Promise<LoopUntilConvergenceResult> — convergent discovery loop. ' +
|
|
29
|
+
'Fans out finders repeatedly, deduplicates against seen-set (NOT confirmed-set), ' +
|
|
30
|
+
'stops after N consecutive dry rounds or maxRounds. Optionally verifies each finding.\n' +
|
|
31
|
+
'- phase(title: string): void — start a new progress group\n' +
|
|
32
|
+
'- log(message: string): void — emit progress message\n' +
|
|
33
|
+
'- args: any — verbatim args passed to Workflow tool\n' +
|
|
34
|
+
'- budget: {total, spent(), remaining()} — token budget tracking\n\n' +
|
|
35
|
+
'## Topology Selection Guide\n\n' +
|
|
36
|
+
'DEFAULT TO pipeline(). Only reach for a barrier (parallel between stages) when you genuinely ' +
|
|
37
|
+
'need ALL prior-stage results together.\n\n' +
|
|
38
|
+
'A barrier is correct ONLY when stage N needs cross-item context from all of stage N-1: ' +
|
|
39
|
+
'dedup/merge across the full result set, early-exit if total count is zero, cross-finding comparison.\n\n' +
|
|
40
|
+
'A barrier is NOT justified by: flatten/map/filter (do it inside a pipeline stage), ' +
|
|
41
|
+
'conceptually separate stages, cleaner code — barrier latency is real and measurable.\n\n' +
|
|
42
|
+
'- Diamond (fan-out → reduce → synthesize): market scans, audits, research\n' +
|
|
43
|
+
'- Pipeline (no barrier): each item flows independently — DEFAULT\n' +
|
|
44
|
+
'- Loop-until-convergence: unknown-size discovery (bugs, vulnerabilities, edge cases)\n' +
|
|
45
|
+
'- Judge panel: multiple competing approaches, pick best + graft runner-ups\n' +
|
|
46
|
+
'- Verifier-on-edge: quality gates before results reach downstream\n\n' +
|
|
47
|
+
'## Critical Rules\n\n' +
|
|
48
|
+
'- EDGE LOGIC IS FREE: flatten, dedupe, filter in plain JavaScript — NOT agent calls. ' +
|
|
49
|
+
'results.flatMap(...) and a Set are deterministic, instant, zero tokens.\n' +
|
|
50
|
+
'- seen-set dedup for loops, NOT confirmed-set — rejected findings would otherwise revive every round.\n' +
|
|
51
|
+
'- Each node should have bounded input, validated output (schema), and one clear purpose.\n' +
|
|
52
|
+
'- Model tiering: use cheaper models for repetitive extraction/classification nodes, ' +
|
|
53
|
+
'expensive models for synthesis/judgment nodes.\n\n' +
|
|
54
|
+
'## Script Format\n\n' +
|
|
55
|
+
'Every script MUST begin with: export const meta = { name, description, phases: [{title, detail}] }\n' +
|
|
56
|
+
'The meta object must be a PURE LITERAL — no variables, function calls, or template interpolation.\n' +
|
|
57
|
+
'Use the SAME phase titles in meta.phases as in phase() calls.',
|
|
11
58
|
category: 'agent',
|
|
12
59
|
permission: 'ask',
|
|
13
60
|
parameters: {
|
|
@@ -60,6 +107,23 @@ export const workflowTool: ToolDefinition = {
|
|
|
60
107
|
resumeFromRunId,
|
|
61
108
|
)
|
|
62
109
|
|
|
110
|
+
// Persist last-run state for /workflow save
|
|
111
|
+
try {
|
|
112
|
+
const { existsSync, mkdirSync, writeFileSync } = await import('node:fs')
|
|
113
|
+
const { join } = await import('node:path')
|
|
114
|
+
const workflowsDir = join(process.cwd(), '.claude', 'workflows')
|
|
115
|
+
if (!existsSync(workflowsDir)) {
|
|
116
|
+
mkdirSync(workflowsDir, { recursive: true })
|
|
117
|
+
}
|
|
118
|
+
writeFileSync(
|
|
119
|
+
join(workflowsDir, '.last-run.json'),
|
|
120
|
+
JSON.stringify({ runId, script, timestamp: new Date().toISOString() }),
|
|
121
|
+
'utf-8',
|
|
122
|
+
)
|
|
123
|
+
} catch {
|
|
124
|
+
// best-effort — don't fail the workflow if state persistence fails
|
|
125
|
+
}
|
|
126
|
+
|
|
63
127
|
let content = `Workflow ${runId} completed.\n\n`
|
|
64
128
|
if (resumeFromRunId) {
|
|
65
129
|
content += `Cache: ${cacheHits} hits · ${cacheMisses} live\n\n`
|
package/src/ui/commands.ts
CHANGED
|
@@ -2388,6 +2388,111 @@ const workflowsCmd: CommandHandler = async () => {
|
|
|
2388
2388
|
return { content: lines.join('\n') }
|
|
2389
2389
|
}
|
|
2390
2390
|
|
|
2391
|
+
// ═══════════════════════════════════════════════════════════════
|
|
2392
|
+
// /workflow <task> — auto-generate + execute
|
|
2393
|
+
// ═══════════════════════════════════════════════════════════════
|
|
2394
|
+
|
|
2395
|
+
const workflowSaveCmd = async (name: string): Promise<CommandResult> => {
|
|
2396
|
+
const { existsSync, mkdirSync, writeFileSync, readFileSync } = await import('node:fs')
|
|
2397
|
+
const { join } = await import('node:path')
|
|
2398
|
+
|
|
2399
|
+
const targetDir = join(process.cwd(), '.claude', 'workflows')
|
|
2400
|
+
if (!existsSync(targetDir)) {
|
|
2401
|
+
mkdirSync(targetDir, { recursive: true })
|
|
2402
|
+
}
|
|
2403
|
+
|
|
2404
|
+
// Read the last-run state persisted by the Workflow tool
|
|
2405
|
+
const stateFile = join(targetDir, '.last-run.json')
|
|
2406
|
+
if (!existsSync(stateFile)) {
|
|
2407
|
+
return { content: 'No recent workflow run found. Run a workflow first with /workflow <task>.' }
|
|
2408
|
+
}
|
|
2409
|
+
|
|
2410
|
+
try {
|
|
2411
|
+
const state = JSON.parse(readFileSync(stateFile, 'utf-8'))
|
|
2412
|
+
const script = state.script as string
|
|
2413
|
+
if (!script) {
|
|
2414
|
+
return { content: 'No script found in last run state.' }
|
|
2415
|
+
}
|
|
2416
|
+
|
|
2417
|
+
const safeName = name.replace(/[^a-zA-Z0-9_-]/g, '-')
|
|
2418
|
+
const scriptPath = join(targetDir, `${safeName}.js`)
|
|
2419
|
+
writeFileSync(scriptPath, script, 'utf-8')
|
|
2420
|
+
|
|
2421
|
+
return {
|
|
2422
|
+
content: `Workflow saved to ${scriptPath}\nUse /workflow run ${safeName} to run it again.`,
|
|
2423
|
+
}
|
|
2424
|
+
} catch (err) {
|
|
2425
|
+
return { content: `Failed to save workflow: ${String(err)}` }
|
|
2426
|
+
}
|
|
2427
|
+
}
|
|
2428
|
+
|
|
2429
|
+
const workflowRunCmd = async (name: string): Promise<CommandResult> => {
|
|
2430
|
+
const { existsSync } = await import('node:fs')
|
|
2431
|
+
const { join } = await import('node:path')
|
|
2432
|
+
|
|
2433
|
+
const safeName = name.replace(/[^a-zA-Z0-9_-]/g, '-')
|
|
2434
|
+
|
|
2435
|
+
const locations = [
|
|
2436
|
+
join(process.cwd(), '.claude', 'workflows'),
|
|
2437
|
+
join(process.env.HOME || '~', '.claude', 'workflows'),
|
|
2438
|
+
]
|
|
2439
|
+
|
|
2440
|
+
for (const loc of locations) {
|
|
2441
|
+
const scriptPath = join(loc, `${safeName}.js`)
|
|
2442
|
+
if (existsSync(scriptPath)) {
|
|
2443
|
+
return {
|
|
2444
|
+
content: '',
|
|
2445
|
+
forwardToAI:
|
|
2446
|
+
`Read the workflow script at ${scriptPath}, then call the Workflow tool with ` +
|
|
2447
|
+
`the file contents as the "script" parameter to execute it. Report the results.`,
|
|
2448
|
+
}
|
|
2449
|
+
}
|
|
2450
|
+
}
|
|
2451
|
+
|
|
2452
|
+
return {
|
|
2453
|
+
content: `Workflow "${safeName}" not found in .claude/workflows/ or ~/.claude/workflows/`,
|
|
2454
|
+
}
|
|
2455
|
+
}
|
|
2456
|
+
|
|
2457
|
+
const workflowAutoCmd: CommandHandler = async (_ctx, args) => {
|
|
2458
|
+
const task = args.join(' ').trim()
|
|
2459
|
+
|
|
2460
|
+
if (!task) {
|
|
2461
|
+
return {
|
|
2462
|
+
content:
|
|
2463
|
+
'Usage: /workflow <task description>\n\n' +
|
|
2464
|
+
'Describes the task, and the AI will generate a workflow script to execute it.\n' +
|
|
2465
|
+
'Examples:\n' +
|
|
2466
|
+
' /workflow audit all routes for missing auth\n' +
|
|
2467
|
+
' /workflow research the impact of React 19 on our codebase\n' +
|
|
2468
|
+
' /workflow find all hardcoded credentials in the codebase\n\n' +
|
|
2469
|
+
'Sub-commands:\n' +
|
|
2470
|
+
' /workflow save <name> — save last successful workflow script\n' +
|
|
2471
|
+
' /workflow run <name> — run a saved workflow by name\n' +
|
|
2472
|
+
' /workflows — list all saved workflow scripts',
|
|
2473
|
+
}
|
|
2474
|
+
}
|
|
2475
|
+
|
|
2476
|
+
// Sub-commands
|
|
2477
|
+
const firstArg = args[0] || ''
|
|
2478
|
+
if (firstArg === 'save' && args[1]) {
|
|
2479
|
+
return workflowSaveCmd(args.slice(1).join(' '))
|
|
2480
|
+
}
|
|
2481
|
+
if (firstArg === 'run' && args[1]) {
|
|
2482
|
+
return workflowRunCmd(args.slice(1).join(' '))
|
|
2483
|
+
}
|
|
2484
|
+
|
|
2485
|
+
// Default: forward to AI to generate + execute workflow
|
|
2486
|
+
return {
|
|
2487
|
+
content: '',
|
|
2488
|
+
forwardToAI:
|
|
2489
|
+
`Write and execute a workflow script for this task: ${task}\n\n` +
|
|
2490
|
+
`Use the Workflow tool to execute the generated script. ` +
|
|
2491
|
+
`After the workflow completes, summarize the results and offer to save the script ` +
|
|
2492
|
+
`with /workflow save <name> if it is reusable.`,
|
|
2493
|
+
}
|
|
2494
|
+
}
|
|
2495
|
+
|
|
2391
2496
|
// ═══════════════════════════════════════════════════════════════
|
|
2392
2497
|
// Permissions
|
|
2393
2498
|
// ═══════════════════════════════════════════════════════════════
|
|
@@ -4318,6 +4423,7 @@ registry.set('/tasks', tasksCmd)
|
|
|
4318
4423
|
registry.set('/diff', diffCmd)
|
|
4319
4424
|
registry.set('/loop', loopCmd)
|
|
4320
4425
|
registry.set('/no-plan', noPlanCmd)
|
|
4426
|
+
registry.set('/workflow', workflowAutoCmd)
|
|
4321
4427
|
registry.set('/workflows', workflowsCmd)
|
|
4322
4428
|
registry.set('/review', reviewCmd)
|
|
4323
4429
|
registry.set('/pr-comments', prCommentsCmd)
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import { parallel } from './parallel'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* loopUntilConvergence() — workflow primitive for iterative discovery with
|
|
5
|
+
* deduplication, adversarial verification, and convergence detection.
|
|
6
|
+
*
|
|
7
|
+
* Orchestrates multiple "finders" over successive rounds until no new items
|
|
8
|
+
* are discovered for `dryRounds` consecutive rounds (convergence), or
|
|
9
|
+
* `maxRounds` is reached (cutoff).
|
|
10
|
+
*
|
|
11
|
+
* Each item is keyed by `keyFn` for deduplication across rounds. If a
|
|
12
|
+
* `verify` function is provided, only items that survive verification
|
|
13
|
+
* enter the confirmed set — but all seen items (even those that fail
|
|
14
|
+
* verification) are tracked in the seen-set to prevent re-discovery.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
export interface VerifyVote {
|
|
18
|
+
real: boolean
|
|
19
|
+
reason: string
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export interface VerifyResult<T = unknown> {
|
|
23
|
+
finding: T
|
|
24
|
+
survives: boolean
|
|
25
|
+
votes: VerifyVote[]
|
|
26
|
+
score: number
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export interface LoopUntilConvergenceResult<T> {
|
|
30
|
+
confirmed: T[]
|
|
31
|
+
totalSeen: number
|
|
32
|
+
rounds: number
|
|
33
|
+
converged: boolean
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface LoopUntilConvergenceOpts<T> {
|
|
37
|
+
finders: Array<() => Promise<{ items: T[] } | null>>
|
|
38
|
+
keyFn: (item: T) => string
|
|
39
|
+
verify?: (item: T) => Promise<VerifyResult<T>>
|
|
40
|
+
dryRounds?: number
|
|
41
|
+
maxRounds?: number
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export async function loopUntilConvergence<T>(
|
|
45
|
+
opts: LoopUntilConvergenceOpts<T>,
|
|
46
|
+
): Promise<LoopUntilConvergenceResult<T>> {
|
|
47
|
+
const dryRounds = opts.dryRounds ?? 2
|
|
48
|
+
const maxRounds = opts.maxRounds ?? 20
|
|
49
|
+
|
|
50
|
+
const seen = new Set<string>()
|
|
51
|
+
const confirmed: T[] = []
|
|
52
|
+
let dry = 0
|
|
53
|
+
let rounds = 0
|
|
54
|
+
|
|
55
|
+
while (dry < dryRounds && rounds < maxRounds) {
|
|
56
|
+
rounds++
|
|
57
|
+
|
|
58
|
+
// FAN OUT: all finders run in parallel
|
|
59
|
+
const raw = await parallel(opts.finders.map((f) => () => f()))
|
|
60
|
+
|
|
61
|
+
// EDGE LOGIC: flatMap + dedup (pure JS, zero tokens)
|
|
62
|
+
const items: T[] = []
|
|
63
|
+
for (const result of raw) {
|
|
64
|
+
if (result && result.items) {
|
|
65
|
+
items.push(...result.items)
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// Dedup against SEEN set, not confirmed
|
|
70
|
+
const fresh = items.filter((item) => {
|
|
71
|
+
const key = opts.keyFn(item)
|
|
72
|
+
if (seen.has(key)) return false
|
|
73
|
+
seen.add(key)
|
|
74
|
+
return true
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
if (fresh.length === 0) {
|
|
78
|
+
dry++ // no new unique items → trending toward convergence
|
|
79
|
+
continue
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
dry = 0 // new items found → reset dry counter
|
|
83
|
+
|
|
84
|
+
// VERIFY: optional quality gate
|
|
85
|
+
if (opts.verify) {
|
|
86
|
+
const judged = await parallel(fresh.map((item) => () => opts.verify!(item)))
|
|
87
|
+
for (const j of judged) {
|
|
88
|
+
if (j && j.survives) {
|
|
89
|
+
confirmed.push(j.finding as T)
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
} else {
|
|
93
|
+
confirmed.push(...fresh)
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
confirmed,
|
|
99
|
+
totalSeen: seen.size,
|
|
100
|
+
rounds,
|
|
101
|
+
converged: dry >= dryRounds,
|
|
102
|
+
}
|
|
103
|
+
}
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* verify() and judge() — workflow primitives for adversarial verification,
|
|
3
|
+
* multi-perspective review, and multi-judge evaluation.
|
|
4
|
+
*
|
|
5
|
+
* verify() supports 3 modes:
|
|
6
|
+
* - adversarial: N skeptics try to refute the finding (majority wins)
|
|
7
|
+
* - perspective: N lenses each evaluate from a specific angle (at least 1 confirms)
|
|
8
|
+
* - consensus: N voters must unanimously agree
|
|
9
|
+
*
|
|
10
|
+
* judge() evaluates N attempts by M judges, computes average scores,
|
|
11
|
+
* picks the winner, and optionally synthesizes a final result.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { workflowAgent } from './agent'
|
|
15
|
+
import { parallel } from './parallel'
|
|
16
|
+
import type { WorkflowAgentOpts } from './agent'
|
|
17
|
+
|
|
18
|
+
// Agent function signature matching the sandbox-injected pattern:
|
|
19
|
+
// (prompt, opts?) → result. In production, the workflow runtime binds
|
|
20
|
+
// ProviderRegistry + ToolRegistry into a function of this shape and
|
|
21
|
+
// injects it via _mockAgent.
|
|
22
|
+
type AgentFn = (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
23
|
+
|
|
24
|
+
// ── Types ──
|
|
25
|
+
|
|
26
|
+
export interface VerifyResult {
|
|
27
|
+
finding: unknown
|
|
28
|
+
survives: boolean
|
|
29
|
+
votes: Array<{ real: boolean; reason: string; lens?: string }>
|
|
30
|
+
score: number
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export type VerifyMode = 'adversarial' | 'perspective' | 'consensus'
|
|
34
|
+
|
|
35
|
+
export interface VerifyOpts {
|
|
36
|
+
mode: VerifyMode
|
|
37
|
+
skeptics?: number // adversarial: default 3
|
|
38
|
+
lenses?: string[] // perspective: e.g. ['correctness', 'security']
|
|
39
|
+
voters?: number // consensus: default 3
|
|
40
|
+
threshold?: number
|
|
41
|
+
schema: Record<string, unknown>
|
|
42
|
+
/** Test-only: inject a mock agent function. When not provided, falls back to workflowAgent. */
|
|
43
|
+
_mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
interface VerdictVote {
|
|
47
|
+
real: boolean
|
|
48
|
+
reason: string
|
|
49
|
+
lens?: string
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export interface JudgeResult {
|
|
53
|
+
winner: unknown
|
|
54
|
+
winnerIndex: number
|
|
55
|
+
scores: Array<{
|
|
56
|
+
attemptIndex: number
|
|
57
|
+
judgeIndex: number
|
|
58
|
+
criteria: Record<string, number>
|
|
59
|
+
total: number
|
|
60
|
+
notes: string
|
|
61
|
+
}>
|
|
62
|
+
synthesis?: string
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export interface JudgeOpts {
|
|
66
|
+
criteria: string[]
|
|
67
|
+
judges?: number // default: 3
|
|
68
|
+
synthesize?: boolean // default: true
|
|
69
|
+
schema: Record<string, unknown>
|
|
70
|
+
/** Test-only: inject a mock agent function. */
|
|
71
|
+
_mockAgent?: (prompt: string, opts?: WorkflowAgentOpts) => Promise<unknown>
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ── Helpers ──
|
|
75
|
+
|
|
76
|
+
function defaultThreshold(mode: VerifyMode, total: number): number {
|
|
77
|
+
if (mode === 'consensus') return total // all must agree
|
|
78
|
+
if (mode === 'perspective') return 1 // at least one lens confirms
|
|
79
|
+
return Math.ceil(total / 2) // adversarial: majority
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// ── verify() ──
|
|
83
|
+
|
|
84
|
+
export async function verify(finding: unknown, opts: VerifyOpts): Promise<VerifyResult> {
|
|
85
|
+
// Prefer the test hook (_mockAgent), fall back to workflowAgent.
|
|
86
|
+
// In production sandbox usage, _mockAgent is set to the pre-bound agent
|
|
87
|
+
// function injected by the workflow runtime.
|
|
88
|
+
const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
|
|
89
|
+
const mode = opts.mode
|
|
90
|
+
|
|
91
|
+
const findingStr = JSON.stringify(finding, null, 2)
|
|
92
|
+
const schemaDesc = JSON.stringify(opts.schema)
|
|
93
|
+
|
|
94
|
+
let prompts: Array<{ prompt: string; lens?: string }>
|
|
95
|
+
|
|
96
|
+
switch (mode) {
|
|
97
|
+
case 'adversarial': {
|
|
98
|
+
const count = opts.skeptics ?? 3
|
|
99
|
+
prompts = Array.from({ length: count }, (_, i) => ({
|
|
100
|
+
prompt:
|
|
101
|
+
`You are a skeptical reviewer (skeptic #${i + 1}). Try to REFUTE this finding. Default to real=false if uncertain.\n\n` +
|
|
102
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
103
|
+
}))
|
|
104
|
+
break
|
|
105
|
+
}
|
|
106
|
+
case 'perspective': {
|
|
107
|
+
const lenses = opts.lenses ?? ['correctness']
|
|
108
|
+
prompts = lenses.map((lens) => ({
|
|
109
|
+
prompt:
|
|
110
|
+
`Judge this finding through the "${lens}" lens. Is it valid from this perspective?\n\n` +
|
|
111
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
112
|
+
lens,
|
|
113
|
+
}))
|
|
114
|
+
break
|
|
115
|
+
}
|
|
116
|
+
case 'consensus': {
|
|
117
|
+
const count = opts.voters ?? 3
|
|
118
|
+
prompts = Array.from({ length: count }, () => ({
|
|
119
|
+
prompt:
|
|
120
|
+
`Is this finding correct? Be honest and critical. Vote real=true only if you are fully convinced.\n\n` +
|
|
121
|
+
`Finding:\n${findingStr}\n\nReturn JSON matching this schema:\n${schemaDesc}`,
|
|
122
|
+
}))
|
|
123
|
+
break
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// Fan out all agent calls concurrently
|
|
128
|
+
const rawVotes = await parallel(
|
|
129
|
+
prompts.map(
|
|
130
|
+
(p) => () => agentFn(p.prompt, { schema: opts.schema as WorkflowAgentOpts['schema'] }),
|
|
131
|
+
),
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
// Filter out failed agents (null) and build vote objects
|
|
135
|
+
const votes: VerdictVote[] = []
|
|
136
|
+
for (let i = 0; i < rawVotes.length; i++) {
|
|
137
|
+
const v = rawVotes[i]
|
|
138
|
+
if (v && typeof v === 'object') {
|
|
139
|
+
const vote = v as Record<string, unknown>
|
|
140
|
+
votes.push({
|
|
141
|
+
real: Boolean(vote.real),
|
|
142
|
+
reason: String(vote.reason ?? ''),
|
|
143
|
+
lens: prompts[i]?.lens,
|
|
144
|
+
})
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const threshold = opts.threshold ?? defaultThreshold(mode, votes.length)
|
|
149
|
+
const realCount = votes.filter((v) => v.real).length
|
|
150
|
+
const survives = realCount >= threshold
|
|
151
|
+
|
|
152
|
+
return {
|
|
153
|
+
finding,
|
|
154
|
+
survives,
|
|
155
|
+
votes,
|
|
156
|
+
score: votes.length > 0 ? realCount / votes.length : 0,
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// ── judge() ──
|
|
161
|
+
|
|
162
|
+
export async function judge(attempts: unknown[], opts: JudgeOpts): Promise<JudgeResult> {
|
|
163
|
+
// Prefer the test hook, fall back to workflowAgent.
|
|
164
|
+
const agentFn: AgentFn = opts._mockAgent ?? (workflowAgent as unknown as AgentFn)
|
|
165
|
+
const judgeCount = opts.judges ?? 3
|
|
166
|
+
const schemaDesc = JSON.stringify(opts.schema)
|
|
167
|
+
|
|
168
|
+
// Phase 1: each judge scores each attempt
|
|
169
|
+
interface ScoreEntry {
|
|
170
|
+
attemptIndex: number
|
|
171
|
+
judgeIndex: number
|
|
172
|
+
criteria: Record<string, number>
|
|
173
|
+
total: number
|
|
174
|
+
notes: string
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// Build the cross-product of judges × attempts
|
|
178
|
+
const scorePrompts: Array<{
|
|
179
|
+
attempt: unknown
|
|
180
|
+
attemptIndex: number
|
|
181
|
+
judgeIndex: number
|
|
182
|
+
}> = []
|
|
183
|
+
for (let ji = 0; ji < judgeCount; ji++) {
|
|
184
|
+
for (let ai = 0; ai < attempts.length; ai++) {
|
|
185
|
+
scorePrompts.push({
|
|
186
|
+
attempt: attempts[ai],
|
|
187
|
+
attemptIndex: ai,
|
|
188
|
+
judgeIndex: ji,
|
|
189
|
+
})
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// Fan out all scoring calls concurrently
|
|
194
|
+
const rawScores = await parallel(
|
|
195
|
+
scorePrompts.map(
|
|
196
|
+
(sp) => () =>
|
|
197
|
+
agentFn(
|
|
198
|
+
`You are judge #${sp.judgeIndex + 1}. Score this attempt against the criteria: ${opts.criteria.join(', ')}.\n\n` +
|
|
199
|
+
`Attempt:\n${JSON.stringify(sp.attempt, null, 2)}\n\n` +
|
|
200
|
+
`Return JSON matching this schema:\n${schemaDesc}`,
|
|
201
|
+
{ schema: opts.schema as WorkflowAgentOpts['schema'] },
|
|
202
|
+
),
|
|
203
|
+
),
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
// Parse and validate each score result
|
|
207
|
+
const scores: ScoreEntry[] = []
|
|
208
|
+
for (let i = 0; i < rawScores.length; i++) {
|
|
209
|
+
const raw = rawScores[i]
|
|
210
|
+
const sp = scorePrompts[i]!
|
|
211
|
+
if (raw && typeof raw === 'object') {
|
|
212
|
+
const obj = raw as Record<string, unknown>
|
|
213
|
+
const criteriaObj = (obj.scores as Record<string, number>) ?? {}
|
|
214
|
+
const total = Object.values(criteriaObj).reduce(
|
|
215
|
+
(sum, v) => sum + (typeof v === 'number' ? v : 0),
|
|
216
|
+
0,
|
|
217
|
+
)
|
|
218
|
+
scores.push({
|
|
219
|
+
attemptIndex: sp.attemptIndex,
|
|
220
|
+
judgeIndex: sp.judgeIndex,
|
|
221
|
+
criteria: criteriaObj,
|
|
222
|
+
total,
|
|
223
|
+
notes: String(obj.notes ?? ''),
|
|
224
|
+
})
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// Compute winner: highest average total across judges
|
|
229
|
+
const attemptTotals = new Map<number, number>()
|
|
230
|
+
const attemptCounts = new Map<number, number>()
|
|
231
|
+
for (const s of scores) {
|
|
232
|
+
attemptTotals.set(s.attemptIndex, (attemptTotals.get(s.attemptIndex) ?? 0) + s.total)
|
|
233
|
+
attemptCounts.set(s.attemptIndex, (attemptCounts.get(s.attemptIndex) ?? 0) + 1)
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
let winnerIndex = 0
|
|
237
|
+
let bestAvg = -Infinity
|
|
238
|
+
for (const [idx, total] of attemptTotals) {
|
|
239
|
+
const count = attemptCounts.get(idx) ?? 1
|
|
240
|
+
const avg = total / count
|
|
241
|
+
if (avg > bestAvg) {
|
|
242
|
+
bestAvg = avg
|
|
243
|
+
winnerIndex = idx
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// Phase 2: optional synthesis
|
|
248
|
+
let synthesis: string | undefined
|
|
249
|
+
if (opts.synthesize !== false) {
|
|
250
|
+
const winner = attempts[winnerIndex]
|
|
251
|
+
const runnerUps = attempts
|
|
252
|
+
.map((a, i) => ({ attempt: a, index: i }))
|
|
253
|
+
.filter((e) => e.index !== winnerIndex)
|
|
254
|
+
|
|
255
|
+
const synthPrompt =
|
|
256
|
+
`Synthesize the final result from the WINNING approach, grafting the best ideas from runner-ups.\n\n` +
|
|
257
|
+
`WINNER:\n${JSON.stringify(winner, null, 2)}\n\n` +
|
|
258
|
+
`RUNNER-UPS:\n${JSON.stringify(runnerUps, null, 2)}\n\n` +
|
|
259
|
+
`Provide a comprehensive synthesis combining the winner's structure with the best elements from other approaches.`
|
|
260
|
+
|
|
261
|
+
const synthResult = await agentFn(synthPrompt)
|
|
262
|
+
if (typeof synthResult === 'string') {
|
|
263
|
+
synthesis = synthResult
|
|
264
|
+
} else if (synthResult && typeof synthResult === 'object' && 'synthesis' in synthResult) {
|
|
265
|
+
synthesis = String((synthResult as Record<string, unknown>).synthesis)
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
return {
|
|
270
|
+
winner: attempts[winnerIndex],
|
|
271
|
+
winnerIndex,
|
|
272
|
+
scores,
|
|
273
|
+
synthesis,
|
|
274
|
+
}
|
|
275
|
+
}
|
package/src/workflow/runtime.ts
CHANGED
|
@@ -5,6 +5,8 @@ import { workflowAgent } from './primitives/agent'
|
|
|
5
5
|
import { parallel } from './primitives/parallel'
|
|
6
6
|
import { pipeline } from './primitives/pipeline'
|
|
7
7
|
import { phase as phasePrimitive } from './primitives/phase'
|
|
8
|
+
import { verify, judge } from './primitives/verify'
|
|
9
|
+
import { loopUntilConvergence } from './primitives/loop'
|
|
8
10
|
import type { ProviderRegistry } from '../providers/registry'
|
|
9
11
|
import type { QueryEngine } from '../core/engine'
|
|
10
12
|
|
|
@@ -147,6 +149,9 @@ export async function runWorkflow(
|
|
|
147
149
|
'agent',
|
|
148
150
|
'parallel',
|
|
149
151
|
'pipeline',
|
|
152
|
+
'verify',
|
|
153
|
+
'judge',
|
|
154
|
+
'loopUntilConvergence',
|
|
150
155
|
'phase',
|
|
151
156
|
'log',
|
|
152
157
|
'args',
|
|
@@ -154,7 +159,18 @@ export async function runWorkflow(
|
|
|
154
159
|
wrappedScript,
|
|
155
160
|
)
|
|
156
161
|
|
|
157
|
-
const result = await scriptFn(
|
|
162
|
+
const result = await scriptFn(
|
|
163
|
+
agent,
|
|
164
|
+
parallel,
|
|
165
|
+
pipeline,
|
|
166
|
+
verify,
|
|
167
|
+
judge,
|
|
168
|
+
loopUntilConvergence,
|
|
169
|
+
wrappedPhase,
|
|
170
|
+
log,
|
|
171
|
+
args,
|
|
172
|
+
budget,
|
|
173
|
+
)
|
|
158
174
|
|
|
159
175
|
// Count journal entries from state
|
|
160
176
|
const priorEntries = loadJournal(runId)
|