@goodandready/dsh-moa 0.2.25 → 0.2.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/README.ru.md +30 -0
- package/lib/client.js +248 -0
- package/lib/history.js +11 -0
- package/lib/index.js +41 -0
- package/lib/moa-budget.js +97 -0
- package/lib/moa-candidates.js +90 -22
- package/lib/moa-context.js +70 -0
- package/lib/moa-multi-judge.js +301 -0
- package/lib/moa-prompts.js +25 -6
- package/lib/moa-report.js +124 -0
- package/lib/moa-router.js +102 -0
- package/lib/moa-runner.js +131 -152
- package/lib/moa-stream.js +117 -0
- package/lib/moa-test-gate.js +236 -0
- package/lib/routes.js +58 -0
- package/package.json +1 -1
package/lib/moa-candidates.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* DeepSeek Harness Mixture of Agents (MoA) — Candidate Fan-Out & Retry
|
|
3
|
+
* With Per-Candidate Temperature, Temperature Gradient & Local Fallback Resilience
|
|
3
4
|
*/
|
|
4
5
|
|
|
5
6
|
import { bestEffort } from './best-effort.js'
|
|
@@ -29,7 +30,8 @@ export async function callWithTransientRetry(callLlmFn, callArgs, maxRetries = 0
|
|
|
29
30
|
}
|
|
30
31
|
|
|
31
32
|
/**
|
|
32
|
-
* Dispatches queries to all reference models in parallel with transient retry
|
|
33
|
+
* Dispatches queries to all reference models in parallel with transient retry,
|
|
34
|
+
* per-candidate temperature / gradient exploration, local fallback, and quorum mitigation.
|
|
33
35
|
*/
|
|
34
36
|
export async function runReferencesParallel(references, messages, options = {}, callLlm, onProgress) {
|
|
35
37
|
if (!Array.isArray(references) || references.length === 0) {
|
|
@@ -87,6 +89,14 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
87
89
|
: ''
|
|
88
90
|
const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
|
|
89
91
|
|
|
92
|
+
// Feature 7: Per-candidate temperature or temperature gradient
|
|
93
|
+
let effectiveTemperature = options.temperature ?? 0.6
|
|
94
|
+
if (typeof slot.temperature === 'number' && slot.temperature >= 0) {
|
|
95
|
+
effectiveTemperature = slot.temperature
|
|
96
|
+
} else if (options.temperature_gradient_enabled && total > 1) {
|
|
97
|
+
effectiveTemperature = Number((0.2 + (0.7 * i) / (total - 1)).toFixed(2))
|
|
98
|
+
}
|
|
99
|
+
|
|
90
100
|
const runOne = async () => {
|
|
91
101
|
try {
|
|
92
102
|
const callPromise = callWithTransientRetry(
|
|
@@ -95,7 +105,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
95
105
|
provider: slot.provider,
|
|
96
106
|
model: slot.model,
|
|
97
107
|
messages: slotMessages,
|
|
98
|
-
temperature:
|
|
108
|
+
temperature: effectiveTemperature,
|
|
99
109
|
maxTokens: options.maxTokens ?? 4096,
|
|
100
110
|
timeoutMs,
|
|
101
111
|
signal: abortControllers[i].signal,
|
|
@@ -124,38 +134,96 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
124
134
|
const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
|
|
125
135
|
|
|
126
136
|
finishedCount++
|
|
127
|
-
if (typeof onProgress === 'function') {
|
|
128
|
-
const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
|
|
129
|
-
onProgress(`✅ *Candidate ${i + 1}/${total} (${label}) finished${costStr}*\n`)
|
|
130
|
-
}
|
|
131
|
-
|
|
132
137
|
results[i] = {
|
|
133
138
|
index: i + 1,
|
|
134
139
|
slot,
|
|
135
140
|
label,
|
|
136
141
|
role_persona: role,
|
|
142
|
+
temperature: effectiveTemperature,
|
|
137
143
|
text,
|
|
138
144
|
usage: costInfo,
|
|
139
145
|
costUsd: costInfo.costUsd,
|
|
140
146
|
ok: true,
|
|
141
147
|
}
|
|
142
|
-
|
|
143
|
-
finishedCount++
|
|
144
|
-
const errMsg = err?.message || String(err)
|
|
145
|
-
console.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
|
|
148
|
+
|
|
146
149
|
if (typeof onProgress === 'function') {
|
|
147
|
-
|
|
150
|
+
const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
|
|
151
|
+
const currentTotalCost = results.reduce((sum, r) => sum + (r?.costUsd || 0), 0)
|
|
152
|
+
onProgress(`✅ *Candidate ${i + 1}/${total} (${label}, T=${effectiveTemperature}) finished${costStr}* | Live total: ~\$${currentTotalCost.toFixed(4)}\n`)
|
|
148
153
|
}
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
154
|
+
} catch (err) {
|
|
155
|
+
// Feature 10: Local fallback models if online candidate fails
|
|
156
|
+
let fallbackRecovered = false
|
|
157
|
+
if (options.local_fallback_enabled && Array.isArray(options.local_fallback_models) && options.local_fallback_models.length > 0) {
|
|
158
|
+
for (const fbSlot of options.local_fallback_models) {
|
|
159
|
+
const fbLabel = slotLabel(fbSlot)
|
|
160
|
+
try {
|
|
161
|
+
if (typeof onProgress === 'function') {
|
|
162
|
+
onProgress(`🔄 *Candidate ${i + 1} (${label}) failed, attempting local fallback to ${fbLabel}...*\n`)
|
|
163
|
+
}
|
|
164
|
+
const fbRes = await callWithTransientRetry(
|
|
165
|
+
callLlm,
|
|
166
|
+
{
|
|
167
|
+
provider: fbSlot.provider,
|
|
168
|
+
model: fbSlot.model,
|
|
169
|
+
messages: slotMessages,
|
|
170
|
+
temperature: effectiveTemperature,
|
|
171
|
+
maxTokens: options.maxTokens ?? 4096,
|
|
172
|
+
timeoutMs,
|
|
173
|
+
signal: abortControllers[i].signal,
|
|
174
|
+
},
|
|
175
|
+
0
|
|
176
|
+
)
|
|
177
|
+
const fbText = typeof fbRes === 'string' ? fbRes : (fbRes?.content || fbRes?.text || '')
|
|
178
|
+
const fbUsage = (typeof fbRes === 'object' && fbRes?.usage) ? fbRes.usage : {
|
|
179
|
+
inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
|
|
180
|
+
outputTokens: Math.max(1, Math.round(fbText.length / 4)),
|
|
181
|
+
}
|
|
182
|
+
const fbCost = estimateTokenCost(fbSlot, fbUsage, options.prices)
|
|
183
|
+
finishedCount++
|
|
184
|
+
results[i] = {
|
|
185
|
+
index: i + 1,
|
|
186
|
+
slot: fbSlot,
|
|
187
|
+
label: `${fbLabel} [fallback for ${label}]`,
|
|
188
|
+
role_persona: role,
|
|
189
|
+
temperature: effectiveTemperature,
|
|
190
|
+
text: fbText,
|
|
191
|
+
usage: fbCost,
|
|
192
|
+
costUsd: fbCost.costUsd,
|
|
193
|
+
ok: true,
|
|
194
|
+
was_fallback: true,
|
|
195
|
+
original_model: label,
|
|
196
|
+
}
|
|
197
|
+
fallbackRecovered = true
|
|
198
|
+
if (typeof onProgress === 'function') {
|
|
199
|
+
onProgress(`✅ *Candidate ${i + 1}/${total} recovered using fallback (${fbLabel})*\n`)
|
|
200
|
+
}
|
|
201
|
+
break
|
|
202
|
+
} catch (fbErr) {
|
|
203
|
+
console.warn(`[dsh-moa] Fallback ${fbLabel} also failed:`, fbErr?.message || fbErr)
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (!fallbackRecovered) {
|
|
209
|
+
finishedCount++
|
|
210
|
+
const errMsg = err?.message || String(err)
|
|
211
|
+
console.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
|
|
212
|
+
if (typeof onProgress === 'function') {
|
|
213
|
+
onProgress(`⚠️ *Candidate ${i + 1}/${total} (${label}) error: ${errMsg}*\n`)
|
|
214
|
+
}
|
|
215
|
+
results[i] = {
|
|
216
|
+
index: i + 1,
|
|
217
|
+
slot,
|
|
218
|
+
label,
|
|
219
|
+
role_persona: role,
|
|
220
|
+
temperature: effectiveTemperature,
|
|
221
|
+
text: `[Model ${label} error: ${errMsg}]`,
|
|
222
|
+
usage: { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 },
|
|
223
|
+
costUsd: 0,
|
|
224
|
+
ok: false,
|
|
225
|
+
error: errMsg,
|
|
226
|
+
}
|
|
159
227
|
}
|
|
160
228
|
} finally {
|
|
161
229
|
notifyFinished()
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Multi-Turn Conversation Memory & Context Pruning
|
|
3
|
+
* Feature 8 (Multi-Turn Continuity)
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Extracts concise prior turn baseline from previous conversation messages.
|
|
8
|
+
*/
|
|
9
|
+
export function extractPriorTurnBaseline(messages = []) {
|
|
10
|
+
if (!Array.isArray(messages) || messages.length === 0) return null
|
|
11
|
+
|
|
12
|
+
// Find the last assistant message
|
|
13
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
14
|
+
const msg = messages[i]
|
|
15
|
+
if (msg?.role === 'assistant' && typeof msg?.content === 'string') {
|
|
16
|
+
const text = msg.content.trim()
|
|
17
|
+
if (text.length > 0) {
|
|
18
|
+
// Strip out noisy banners if present
|
|
19
|
+
const cleaned = text
|
|
20
|
+
.replace(/^#+\s*🏆[^\n]+\n/m, '')
|
|
21
|
+
.replace(/<!--\s*moa-metadata[\s\S]*?-->/g, '')
|
|
22
|
+
.trim()
|
|
23
|
+
return cleaned.slice(0, 3500)
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
return null
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Prunes conversation history for candidate fan-out to prevent token ballooning
|
|
32
|
+
* while preserving multi-turn context continuity.
|
|
33
|
+
*/
|
|
34
|
+
export function pruneMultiTurnMessages(messages = [], maxHistoryTokens = 3000) {
|
|
35
|
+
if (!Array.isArray(messages) || messages.length <= 2) {
|
|
36
|
+
return messages || []
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const pruned = []
|
|
40
|
+
const sysMsg = messages.find((m) => m && m.role === 'system')
|
|
41
|
+
if (sysMsg) {
|
|
42
|
+
pruned.push(sysMsg)
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// Get conversation turns excluding system
|
|
46
|
+
const nonSys = messages.filter((m) => m && m.role !== 'system')
|
|
47
|
+
if (nonSys.length === 0) return pruned
|
|
48
|
+
|
|
49
|
+
// Always keep the final 2 turns (last user and prior assistant)
|
|
50
|
+
const recentTurns = nonSys.slice(-4)
|
|
51
|
+
|
|
52
|
+
for (let idx = 0; idx < recentTurns.length; idx++) {
|
|
53
|
+
const msg = recentTurns[idx]
|
|
54
|
+
const isLatest = idx === recentTurns.length - 1
|
|
55
|
+
const content = typeof msg.content === 'string' ? msg.content : JSON.stringify(msg.content)
|
|
56
|
+
|
|
57
|
+
if (isLatest || content.length < 800) {
|
|
58
|
+
pruned.push(msg)
|
|
59
|
+
} else {
|
|
60
|
+
// Prune long prior assistant or user text to protect candidate context
|
|
61
|
+
const truncated = content.slice(0, 750) + '\n\n[...context pruned for token efficiency...]\n' + content.slice(-250)
|
|
62
|
+
pruned.push({
|
|
63
|
+
role: msg.role,
|
|
64
|
+
content: truncated,
|
|
65
|
+
})
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
return pruned
|
|
70
|
+
}
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Multi-Judge Panel Consensus Voting & Composite Hybrid Synthesis
|
|
3
|
+
* Feature 2 (Multi-Judge Panel) & Feature 3 (Composite Block Synthesis)
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import { slotLabel } from './moa-prompts.js'
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Extracts code fences and block-level assets from markdown text.
|
|
10
|
+
*/
|
|
11
|
+
export function extractCodeBlocks(text = '') {
|
|
12
|
+
if (!text || typeof text !== 'string') return []
|
|
13
|
+
const blocks = []
|
|
14
|
+
const regex = /```([a-zA-Z0-9_\-\.\/:]+)?(?:\s+(?:file|path|filename)=([^\s\n]+))?\n([\s\S]*?)```/g
|
|
15
|
+
let match
|
|
16
|
+
while ((match = regex.exec(text)) !== null) {
|
|
17
|
+
const rawTag = (match[1] || '').trim()
|
|
18
|
+
const explicitFile = (match[2] || '').trim()
|
|
19
|
+
let lang = rawTag
|
|
20
|
+
let filename = explicitFile
|
|
21
|
+
|
|
22
|
+
// Handle ```ts:src/app.ts or ```typescript:app.ts
|
|
23
|
+
if (rawTag.includes(':')) {
|
|
24
|
+
const parts = rawTag.split(':')
|
|
25
|
+
lang = parts[0]
|
|
26
|
+
filename = parts.slice(1).join(':')
|
|
27
|
+
} else if (rawTag.includes('/') || rawTag.includes('.')) {
|
|
28
|
+
filename = rawTag
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
blocks.push({
|
|
32
|
+
lang: lang || 'text',
|
|
33
|
+
filename: filename || null,
|
|
34
|
+
code: match[3],
|
|
35
|
+
fullBlock: match[0],
|
|
36
|
+
})
|
|
37
|
+
}
|
|
38
|
+
return blocks
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Builds composite block directives highlighting complementary modules across candidates.
|
|
43
|
+
*/
|
|
44
|
+
export function buildCompositeBlockDirectives(candidates = []) {
|
|
45
|
+
const fileMap = new Map()
|
|
46
|
+
|
|
47
|
+
for (const c of candidates) {
|
|
48
|
+
if (!c.ok || !c.text) continue
|
|
49
|
+
const blocks = extractCodeBlocks(c.text)
|
|
50
|
+
for (const b of blocks) {
|
|
51
|
+
if (b.filename) {
|
|
52
|
+
if (!fileMap.has(b.filename)) {
|
|
53
|
+
fileMap.set(b.filename, [])
|
|
54
|
+
}
|
|
55
|
+
fileMap.get(b.filename).push(c.label || `Candidate ${c.index}`)
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
let fileOverview = ''
|
|
61
|
+
if (fileMap.size > 0) {
|
|
62
|
+
fileOverview = '\nDetected modular file contributions across candidates:\n' +
|
|
63
|
+
Array.from(fileMap.entries())
|
|
64
|
+
.map(([file, contributors]) => `- \`${file}\`: available in ${contributors.join(', ')}`)
|
|
65
|
+
.join('\n') + '\n'
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
return `
|
|
69
|
+
=== COMPOSITE HYBRID SYNTHESIS DIRECTIVE ===
|
|
70
|
+
You must synthesize a unified, best-of-all-worlds composite solution from all candidates.
|
|
71
|
+
Do NOT simply copy one candidate. Instead:
|
|
72
|
+
1. Extract the cleanest and most efficient core logic/algorithm (e.g. from the candidate with the highest reasoning score).
|
|
73
|
+
2. Incorporate defensive error handling, input validation, and edge-case guards.
|
|
74
|
+
3. Retain complete type annotations, utility helpers, and test coverage provided by any candidate.
|
|
75
|
+
4. Ensure all merged modules, functions, and files fit together into a cohesive, non-conflicting solution.
|
|
76
|
+
${fileOverview}============================================
|
|
77
|
+
`
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Parses judge output into structured scores and candidate choice.
|
|
82
|
+
*/
|
|
83
|
+
export function parseJudgeEvaluation(rawText = '', candidateCount = 2) {
|
|
84
|
+
const scores = {}
|
|
85
|
+
let bestCandidate = 1
|
|
86
|
+
let reasoning = ''
|
|
87
|
+
|
|
88
|
+
const scoreMatches = rawText.matchAll(/Candidate\s*(\d+)[:\s]+(?:Score\s*[:\s]*)?([0-9]+(?:\.[0-9]+)?)/gi)
|
|
89
|
+
for (const m of scoreMatches) {
|
|
90
|
+
const idx = parseInt(m[1], 10)
|
|
91
|
+
const val = parseFloat(m[2])
|
|
92
|
+
if (idx >= 1 && idx <= candidateCount && !Number.isNaN(val)) {
|
|
93
|
+
scores[idx] = Math.min(10, Math.max(0, val))
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const bestMatch = rawText.match(/BEST_CANDIDATE\s*[:\s]+(?:Candidate\s*)?(\d+)/i)
|
|
98
|
+
if (bestMatch) {
|
|
99
|
+
const parsedBest = parseInt(bestMatch[1], 10)
|
|
100
|
+
if (parsedBest >= 1 && parsedBest <= candidateCount) {
|
|
101
|
+
bestCandidate = parsedBest
|
|
102
|
+
}
|
|
103
|
+
} else {
|
|
104
|
+
// Find candidate with max parsed score
|
|
105
|
+
let maxScore = -1
|
|
106
|
+
for (const [idxStr, s] of Object.entries(scores)) {
|
|
107
|
+
if (s > maxScore) {
|
|
108
|
+
maxScore = s
|
|
109
|
+
bestCandidate = parseInt(idxStr, 10)
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const reasonMatch = rawText.match(/CONSENSUS_REASONING\s*[:\s]+([^\n\r]+)/i)
|
|
115
|
+
if (reasonMatch) {
|
|
116
|
+
reasoning = reasonMatch[1].trim()
|
|
117
|
+
} else {
|
|
118
|
+
reasoning = rawText.slice(0, 180).replace(/\n/g, ' ').trim()
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return { scores, bestCandidate, reasoning }
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Executes a panel of judges to reach consensus on candidate solutions.
|
|
126
|
+
*/
|
|
127
|
+
export async function executeMultiJudgePanel({
|
|
128
|
+
judges = [],
|
|
129
|
+
candidates = [],
|
|
130
|
+
userPrompt = '',
|
|
131
|
+
callLlm,
|
|
132
|
+
strategy = 'majority', // 'majority' | 'highest_score' | 'unanimous'
|
|
133
|
+
timeoutMs = 45000,
|
|
134
|
+
signal,
|
|
135
|
+
onProgress,
|
|
136
|
+
}) {
|
|
137
|
+
const activeCandidates = (candidates || []).filter((c) => c && c.ok)
|
|
138
|
+
if (activeCandidates.length === 0) {
|
|
139
|
+
return {
|
|
140
|
+
consensus: false,
|
|
141
|
+
winningCandidateIndex: 1,
|
|
142
|
+
strategy,
|
|
143
|
+
reason: 'no_active_candidates',
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
if (activeCandidates.length === 1) {
|
|
148
|
+
return {
|
|
149
|
+
consensus: true,
|
|
150
|
+
winningCandidateIndex: activeCandidates[0].index,
|
|
151
|
+
winningCandidateLabel: activeCandidates[0].label,
|
|
152
|
+
strategy,
|
|
153
|
+
reason: 'single_active_candidate',
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
const judgeSlots = Array.isArray(judges) && judges.length > 0
|
|
158
|
+
? judges
|
|
159
|
+
: [{ provider: 'default', model: 'aggregator-judge', label: 'Primary Judge' }]
|
|
160
|
+
|
|
161
|
+
const candidateSummary = activeCandidates.map((c) => {
|
|
162
|
+
return `### Candidate ${c.index} (${c.label}):\n${(c.text || '').slice(0, 4000)}`
|
|
163
|
+
}).join('\n\n---\n\n')
|
|
164
|
+
|
|
165
|
+
const judgePrompt = `You are an impartial Expert Code & Logic Judge evaluating multiple AI candidate solutions.
|
|
166
|
+
User Prompt:
|
|
167
|
+
${userPrompt.slice(0, 1000)}
|
|
168
|
+
|
|
169
|
+
Candidate Solutions:
|
|
170
|
+
${candidateSummary}
|
|
171
|
+
|
|
172
|
+
Evaluate the solutions on correctness, architecture, edge cases, and maintainability.
|
|
173
|
+
Rate each candidate on a scale of 1 to 10.
|
|
174
|
+
Pick the single best candidate.
|
|
175
|
+
|
|
176
|
+
Your response MUST follow this exact format:
|
|
177
|
+
SCORES:
|
|
178
|
+
${activeCandidates.map((c) => `Candidate ${c.index}: [1-10]`).join('\n')}
|
|
179
|
+
BEST_CANDIDATE: [index number]
|
|
180
|
+
CONSENSUS_REASONING: [one concise sentence explaining your pick]
|
|
181
|
+
`
|
|
182
|
+
|
|
183
|
+
if (typeof onProgress === 'function') {
|
|
184
|
+
onProgress(`⚖️ *Convening Multi-Judge Panel (${judgeSlots.length} judges, strategy: ${strategy})...*\n`)
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const judgePromises = judgeSlots.map(async (judge, jIdx) => {
|
|
188
|
+
const label = slotLabel(judge)
|
|
189
|
+
try {
|
|
190
|
+
const abortCtrl = new AbortController()
|
|
191
|
+
if (signal) {
|
|
192
|
+
signal.addEventListener('abort', () => abortCtrl.abort(), { once: true })
|
|
193
|
+
}
|
|
194
|
+
const timer = setTimeout(() => abortCtrl.abort(new Error('Judge evaluation timeout')), timeoutMs)
|
|
195
|
+
timer.unref?.()
|
|
196
|
+
|
|
197
|
+
const res = await callLlm({
|
|
198
|
+
provider: judge.provider,
|
|
199
|
+
model: judge.model,
|
|
200
|
+
messages: [
|
|
201
|
+
{ role: 'system', content: 'You are an objective expert judge in an ensemble evaluation system.' },
|
|
202
|
+
{ role: 'user', content: judgePrompt },
|
|
203
|
+
],
|
|
204
|
+
temperature: 0.1,
|
|
205
|
+
maxTokens: 512,
|
|
206
|
+
signal: abortCtrl.signal,
|
|
207
|
+
})
|
|
208
|
+
clearTimeout(timer)
|
|
209
|
+
|
|
210
|
+
const text = typeof res === 'string' ? res : (res?.content || res?.text || '')
|
|
211
|
+
const evalResult = parseJudgeEvaluation(text, candidates.length)
|
|
212
|
+
|
|
213
|
+
return {
|
|
214
|
+
judgeIndex: jIdx + 1,
|
|
215
|
+
judgeLabel: label,
|
|
216
|
+
ok: true,
|
|
217
|
+
...evalResult,
|
|
218
|
+
}
|
|
219
|
+
} catch (err) {
|
|
220
|
+
return {
|
|
221
|
+
judgeIndex: jIdx + 1,
|
|
222
|
+
judgeLabel: label,
|
|
223
|
+
ok: false,
|
|
224
|
+
bestCandidate: 1,
|
|
225
|
+
scores: {},
|
|
226
|
+
reasoning: err?.message || String(err),
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
})
|
|
230
|
+
|
|
231
|
+
const results = await Promise.all(judgePromises)
|
|
232
|
+
const validResults = results.filter((r) => r.ok)
|
|
233
|
+
|
|
234
|
+
// Tally votes and compute average scores
|
|
235
|
+
const votes = {}
|
|
236
|
+
const totalScores = {}
|
|
237
|
+
const scoreCounts = {}
|
|
238
|
+
|
|
239
|
+
for (const r of validResults) {
|
|
240
|
+
votes[r.bestCandidate] = (votes[r.bestCandidate] || 0) + 1
|
|
241
|
+
for (const [cIdx, s] of Object.entries(r.scores || {})) {
|
|
242
|
+
totalScores[cIdx] = (totalScores[cIdx] || 0) + s
|
|
243
|
+
scoreCounts[cIdx] = (scoreCounts[cIdx] || 0) + 1
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
const averageScores = {}
|
|
248
|
+
for (const c of activeCandidates) {
|
|
249
|
+
const cnt = scoreCounts[c.index] || 0
|
|
250
|
+
averageScores[c.index] = cnt > 0 ? Number((totalScores[c.index] / cnt).toFixed(2)) : 5.0
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
let winner = activeCandidates[0].index
|
|
254
|
+
let isUnanimous = false
|
|
255
|
+
|
|
256
|
+
if (strategy === 'highest_score') {
|
|
257
|
+
let highestAvg = -1
|
|
258
|
+
for (const c of activeCandidates) {
|
|
259
|
+
const avg = averageScores[c.index] || 0
|
|
260
|
+
if (avg > highestAvg) {
|
|
261
|
+
highestAvg = avg
|
|
262
|
+
winner = c.index
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
} else {
|
|
266
|
+
// majority or unanimous
|
|
267
|
+
let maxVotes = -1
|
|
268
|
+
for (const c of activeCandidates) {
|
|
269
|
+
const v = votes[c.index] || 0
|
|
270
|
+
if (v > maxVotes) {
|
|
271
|
+
maxVotes = v
|
|
272
|
+
winner = c.index
|
|
273
|
+
} else if (v === maxVotes && (averageScores[c.index] || 0) > (averageScores[winner] || 0)) {
|
|
274
|
+
winner = c.index
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
isUnanimous = (votes[winner] || 0) === validResults.length && validResults.length > 0
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
const winningCand = activeCandidates.find((c) => c.index === winner) || activeCandidates[0]
|
|
281
|
+
|
|
282
|
+
const consensusReport = `Multi-Judge Consensus (${validResults.length}/${judgeSlots.length} judges): ` +
|
|
283
|
+
`Winner is Candidate ${winningCand.index} (${winningCand.label}) with ${votes[winner] || 0} votes, ` +
|
|
284
|
+
`avg score ${averageScores[winner] || 0}/10. Strategy: ${strategy}.`
|
|
285
|
+
|
|
286
|
+
if (typeof onProgress === 'function') {
|
|
287
|
+
onProgress(`🏁 *${consensusReport}*\n`)
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
return {
|
|
291
|
+
consensus: true,
|
|
292
|
+
winningCandidateIndex: winningCand.index,
|
|
293
|
+
winningCandidateLabel: winningCand.label,
|
|
294
|
+
strategy,
|
|
295
|
+
isUnanimous,
|
|
296
|
+
votes,
|
|
297
|
+
averageScores,
|
|
298
|
+
judgesResults: results,
|
|
299
|
+
consensusReport,
|
|
300
|
+
}
|
|
301
|
+
}
|
package/lib/moa-prompts.js
CHANGED
|
@@ -57,6 +57,12 @@ export const ROLE_PERSONA_PROMPTS = {
|
|
|
57
57
|
general: '',
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
+
export const TEST_GATE_DIRECTIVE = `### 🧪 Test Execution Gate Directive:
|
|
61
|
+
Some or all candidates were evaluated by executing real project tests against their sandbox implementations.
|
|
62
|
+
CRITICAL JUDGE DIRECTIVE: Test execution results are empirical ground truth.
|
|
63
|
+
- A candidate whose code passes all tests (PASS) must be strongly favored over candidates whose code causes test failures or exceptions (FAIL).
|
|
64
|
+
- If a candidate with test failures is chosen due to significantly superior architecture, you MUST explicitly identify and repair the failing test/code in your final synthesis deliverable.`
|
|
65
|
+
|
|
60
66
|
export const SYNTAX_CORRECTION_DIRECTIVE = `### ⚠️ Syntax Warning & Auto-Fix Directive:
|
|
61
67
|
Some candidate proposals have detected syntax flaws (annotated with [⚠️ Syntax Warning]).
|
|
62
68
|
CRITICAL JUDGE DIRECTIVE: If a candidate with a syntax flaw presents superior architecture, algorithm, or engineering logic compared to other candidates, DO NOT reject them solely for this syntax flaw!
|
|
@@ -235,10 +241,14 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
|
|
|
235
241
|
const role = r.role_persona || r.slot?.role_persona || r.role || r.slot?.role
|
|
236
242
|
const roleNote = role && role !== 'general' ? ` [Focus: ${role}]` : ''
|
|
237
243
|
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
244
|
+
const testNote = r.testResult ? ` [🧪 Test Gate: ${r.testResult.summary}]` : ''
|
|
238
245
|
const header = isBlind
|
|
239
|
-
? `Candidate ${i + 1}${roleNote}:${syntaxNote}${fileSummary}`
|
|
240
|
-
: `Candidate ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${fileSummary}`
|
|
241
|
-
|
|
246
|
+
? `Candidate ${i + 1}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
247
|
+
: `Candidate ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
248
|
+
const testReport = r.testResult
|
|
249
|
+
? `\n\n[🧪 Test Execution Gate Report]:\n- Status: ${r.testResult.passed ? 'PASSED ✅' : 'FAILED ❌'} (Exit code: ${r.testResult.exitCode})\n- Duration: ${r.testResult.durationMs}ms\n- Output:\n\`\`\`\n${r.testResult.output}\n\`\`\``
|
|
250
|
+
: ''
|
|
251
|
+
return `${header}\n${textContent}${testReport}`
|
|
242
252
|
})
|
|
243
253
|
.join('\n\n')
|
|
244
254
|
|
|
@@ -258,6 +268,7 @@ ${joined}
|
|
|
258
268
|
${ANTIPATTERNS_RUBRIC}
|
|
259
269
|
|
|
260
270
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
271
|
+
${referenceOutputs.some((r) => r.testResult) ? `\n${TEST_GATE_DIRECTIVE}\n` : ''}
|
|
261
272
|
${referenceOutputs.some((r) => r.syntaxWarning) ? `
|
|
262
273
|
${SYNTAX_CORRECTION_DIRECTIVE}
|
|
263
274
|
` : ''}
|
|
@@ -304,10 +315,14 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
|
|
|
304
315
|
const role = r.role_persona || r.slot?.role_persona || r.role || r.slot?.role
|
|
305
316
|
const roleNote = role && role !== 'general' ? ` [Focus: ${role}]` : ''
|
|
306
317
|
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
318
|
+
const testNote = r.testResult ? ` [🧪 Test Gate: ${r.testResult.summary}]` : ''
|
|
307
319
|
const header = isBlind
|
|
308
|
-
? `Reference ${i + 1}${roleNote}:${syntaxNote}${fileSummary}`
|
|
309
|
-
: `Reference ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${fileSummary}`
|
|
310
|
-
|
|
320
|
+
? `Reference ${i + 1}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
321
|
+
: `Reference ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
322
|
+
const testReport = r.testResult
|
|
323
|
+
? `\n\n[🧪 Test Execution Gate Report]:\n- Status: ${r.testResult.passed ? 'PASSED ✅' : 'FAILED ❌'} (Exit code: ${r.testResult.exitCode})\n- Duration: ${r.testResult.durationMs}ms\n- Output:\n\`\`\`\n${r.testResult.output}\n\`\`\``
|
|
324
|
+
: ''
|
|
325
|
+
return `${header}\n${textContent}${testReport}`
|
|
311
326
|
})
|
|
312
327
|
.join('\n\n')
|
|
313
328
|
|
|
@@ -326,7 +341,11 @@ ${joined}
|
|
|
326
341
|
${ANTIPATTERNS_RUBRIC}
|
|
327
342
|
|
|
328
343
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
344
|
+
${referenceOutputs.some((r) => r.testResult) ? `\n${TEST_GATE_DIRECTIVE}\n` : ''}
|
|
329
345
|
${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
|
|
346
|
+
${options.priorTurnBaseline ? `\n### 🔄 Prior Turn Context Baseline:\n${options.priorTurnBaseline}\n` : ''}
|
|
347
|
+
${options.consensusReport ? `\n### ⚖️ Multi-Judge Consensus Advisory:\n${options.consensusReport}\n` : ''}
|
|
348
|
+
${options.compositeMergeDirective ? `\n${options.compositeMergeDirective}\n` : ''}
|
|
330
349
|
Instructions:
|
|
331
350
|
Your response MUST be structured into three clear parts:
|
|
332
351
|
|