@goodandready/dsh-moa 0.2.26 → 0.2.29

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,7 @@
1
+ import { logger } from './logger.js'
1
2
  /**
2
3
  * DeepSeek Harness Mixture of Agents (MoA) — Candidate Fan-Out & Retry
4
+ * With Per-Candidate Temperature, Temperature Gradient & Local Fallback Resilience
3
5
  */
4
6
 
5
7
  import { bestEffort } from './best-effort.js'
@@ -29,7 +31,8 @@ export async function callWithTransientRetry(callLlmFn, callArgs, maxRetries = 0
29
31
  }
30
32
 
31
33
  /**
32
- * Dispatches queries to all reference models in parallel with transient retry and quorum straggler mitigation.
34
+ * Dispatches queries to all reference models in parallel with transient retry,
35
+ * per-candidate temperature / gradient exploration, local fallback, and quorum mitigation.
33
36
  */
34
37
  export async function runReferencesParallel(references, messages, options = {}, callLlm, onProgress) {
35
38
  if (!Array.isArray(references) || references.length === 0) {
@@ -87,6 +90,14 @@ export async function runReferencesParallel(references, messages, options = {},
87
90
  : ''
88
91
  const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
89
92
 
93
+ // Feature 7: Per-candidate temperature or temperature gradient
94
+ let effectiveTemperature = options.temperature ?? 0.6
95
+ if (typeof slot.temperature === 'number' && slot.temperature >= 0) {
96
+ effectiveTemperature = slot.temperature
97
+ } else if (options.temperature_gradient_enabled && total > 1) {
98
+ effectiveTemperature = Number((0.2 + (0.7 * i) / (total - 1)).toFixed(2))
99
+ }
100
+
90
101
  const runOne = async () => {
91
102
  try {
92
103
  const callPromise = callWithTransientRetry(
@@ -95,7 +106,7 @@ export async function runReferencesParallel(references, messages, options = {},
95
106
  provider: slot.provider,
96
107
  model: slot.model,
97
108
  messages: slotMessages,
98
- temperature: options.temperature ?? 0.6,
109
+ temperature: effectiveTemperature,
99
110
  maxTokens: options.maxTokens ?? 4096,
100
111
  timeoutMs,
101
112
  signal: abortControllers[i].signal,
@@ -124,38 +135,96 @@ export async function runReferencesParallel(references, messages, options = {},
124
135
  const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
125
136
 
126
137
  finishedCount++
127
- if (typeof onProgress === 'function') {
128
- const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
129
- onProgress(`✅ *Candidate ${i + 1}/${total} (${label}) finished${costStr}*\n`)
130
- }
131
-
132
138
  results[i] = {
133
139
  index: i + 1,
134
140
  slot,
135
141
  label,
136
142
  role_persona: role,
143
+ temperature: effectiveTemperature,
137
144
  text,
138
145
  usage: costInfo,
139
146
  costUsd: costInfo.costUsd,
140
147
  ok: true,
141
148
  }
142
- } catch (err) {
143
- finishedCount++
144
- const errMsg = err?.message || String(err)
145
- console.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
149
+
146
150
  if (typeof onProgress === 'function') {
147
- onProgress(`⚠️ *Candidate ${i + 1}/${total} (${label}) error: ${errMsg}*\n`)
151
+ const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
152
+ const currentTotalCost = results.reduce((sum, r) => sum + (r?.costUsd || 0), 0)
153
+ onProgress(`✅ *Candidate ${i + 1}/${total} (${label}, T=${effectiveTemperature}) finished${costStr}* | Live total: ~\$${currentTotalCost.toFixed(4)}\n`)
148
154
  }
149
- results[i] = {
150
- index: i + 1,
151
- slot,
152
- label,
153
- role_persona: role,
154
- text: `[Model ${label} error: ${errMsg}]`,
155
- usage: { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 },
156
- costUsd: 0,
157
- ok: false,
158
- error: errMsg,
155
+ } catch (err) {
156
+ // Feature 10: Local fallback models if online candidate fails
157
+ let fallbackRecovered = false
158
+ if (options.local_fallback_enabled && Array.isArray(options.local_fallback_models) && options.local_fallback_models.length > 0) {
159
+ for (const fbSlot of options.local_fallback_models) {
160
+ const fbLabel = slotLabel(fbSlot)
161
+ try {
162
+ if (typeof onProgress === 'function') {
163
+ onProgress(`🔄 *Candidate ${i + 1} (${label}) failed, attempting local fallback to ${fbLabel}...*\n`)
164
+ }
165
+ const fbRes = await callWithTransientRetry(
166
+ callLlm,
167
+ {
168
+ provider: fbSlot.provider,
169
+ model: fbSlot.model,
170
+ messages: slotMessages,
171
+ temperature: effectiveTemperature,
172
+ maxTokens: options.maxTokens ?? 4096,
173
+ timeoutMs,
174
+ signal: abortControllers[i].signal,
175
+ },
176
+ 0
177
+ )
178
+ const fbText = typeof fbRes === 'string' ? fbRes : (fbRes?.content || fbRes?.text || '')
179
+ const fbUsage = (typeof fbRes === 'object' && fbRes?.usage) ? fbRes.usage : {
180
+ inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
181
+ outputTokens: Math.max(1, Math.round(fbText.length / 4)),
182
+ }
183
+ const fbCost = estimateTokenCost(fbSlot, fbUsage, options.prices)
184
+ finishedCount++
185
+ results[i] = {
186
+ index: i + 1,
187
+ slot: fbSlot,
188
+ label: `${fbLabel} [fallback for ${label}]`,
189
+ role_persona: role,
190
+ temperature: effectiveTemperature,
191
+ text: fbText,
192
+ usage: fbCost,
193
+ costUsd: fbCost.costUsd,
194
+ ok: true,
195
+ was_fallback: true,
196
+ original_model: label,
197
+ }
198
+ fallbackRecovered = true
199
+ if (typeof onProgress === 'function') {
200
+ onProgress(`✅ *Candidate ${i + 1}/${total} recovered using fallback (${fbLabel})*\n`)
201
+ }
202
+ break
203
+ } catch (fbErr) {
204
+ logger.warn(`[dsh-moa] Fallback ${fbLabel} also failed:`, fbErr?.message || fbErr)
205
+ }
206
+ }
207
+ }
208
+
209
+ if (!fallbackRecovered) {
210
+ finishedCount++
211
+ const errMsg = err?.message || String(err)
212
+ logger.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
213
+ if (typeof onProgress === 'function') {
214
+ onProgress(`⚠️ *Candidate ${i + 1}/${total} (${label}) error: ${errMsg}*\n`)
215
+ }
216
+ results[i] = {
217
+ index: i + 1,
218
+ slot,
219
+ label,
220
+ role_persona: role,
221
+ temperature: effectiveTemperature,
222
+ text: `[Model ${label} error: ${errMsg}]`,
223
+ usage: { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 },
224
+ costUsd: 0,
225
+ ok: false,
226
+ error: errMsg,
227
+ }
159
228
  }
160
229
  } finally {
161
230
  notifyFinished()
@@ -0,0 +1,70 @@
1
+ /**
2
+ * Multi-Turn Conversation Memory & Context Pruning
3
+ * Feature 8 (Multi-Turn Continuity)
4
+ */
5
+
6
+ /**
7
+ * Extracts concise prior turn baseline from previous conversation messages.
8
+ */
9
+ export function extractPriorTurnBaseline(messages = []) {
10
+ if (!Array.isArray(messages) || messages.length === 0) return null
11
+
12
+ // Find the last assistant message
13
+ for (let i = messages.length - 1; i >= 0; i--) {
14
+ const msg = messages[i]
15
+ if (msg?.role === 'assistant' && typeof msg?.content === 'string') {
16
+ const text = msg.content.trim()
17
+ if (text.length > 0) {
18
+ // Strip out noisy banners if present
19
+ const cleaned = text
20
+ .replace(/^#+\s*🏆[^\n]+\n/m, '')
21
+ .replace(/<!--\s*moa-metadata[\s\S]*?-->/g, '')
22
+ .trim()
23
+ return cleaned.slice(0, 3500)
24
+ }
25
+ }
26
+ }
27
+ return null
28
+ }
29
+
30
+ /**
31
+ * Prunes conversation history for candidate fan-out to prevent token ballooning
32
+ * while preserving multi-turn context continuity.
33
+ */
34
+ export function pruneMultiTurnMessages(messages = [], maxHistoryTokens = 3000) {
35
+ if (!Array.isArray(messages) || messages.length <= 2) {
36
+ return messages || []
37
+ }
38
+
39
+ const pruned = []
40
+ const sysMsg = messages.find((m) => m && m.role === 'system')
41
+ if (sysMsg) {
42
+ pruned.push(sysMsg)
43
+ }
44
+
45
+ // Get conversation turns excluding system
46
+ const nonSys = messages.filter((m) => m && m.role !== 'system')
47
+ if (nonSys.length === 0) return pruned
48
+
49
+ // Always keep the final 2 turns (last user and prior assistant)
50
+ const recentTurns = nonSys.slice(-4)
51
+
52
+ for (let idx = 0; idx < recentTurns.length; idx++) {
53
+ const msg = recentTurns[idx]
54
+ const isLatest = idx === recentTurns.length - 1
55
+ const content = typeof msg.content === 'string' ? msg.content : JSON.stringify(msg.content)
56
+
57
+ if (isLatest || content.length < 800) {
58
+ pruned.push(msg)
59
+ } else {
60
+ // Prune long prior assistant or user text to protect candidate context
61
+ const truncated = content.slice(0, 750) + '\n\n[...context pruned for token efficiency...]\n' + content.slice(-250)
62
+ pruned.push({
63
+ role: msg.role,
64
+ content: truncated,
65
+ })
66
+ }
67
+ }
68
+
69
+ return pruned
70
+ }
@@ -0,0 +1,301 @@
1
+ /**
2
+ * Multi-Judge Panel Consensus Voting & Composite Hybrid Synthesis
3
+ * Feature 2 (Multi-Judge Panel) & Feature 3 (Composite Block Synthesis)
4
+ */
5
+
6
+ import { slotLabel } from './moa-prompts.js'
7
+
8
+ /**
9
+ * Extracts code fences and block-level assets from markdown text.
10
+ */
11
+ export function extractCodeBlocks(text = '') {
12
+ if (!text || typeof text !== 'string') return []
13
+ const blocks = []
14
+ const regex = /```([a-zA-Z0-9_\-\.\/:]+)?(?:\s+(?:file|path|filename)=([^\s\n]+))?\n([\s\S]*?)```/g
15
+ let match
16
+ while ((match = regex.exec(text)) !== null) {
17
+ const rawTag = (match[1] || '').trim()
18
+ const explicitFile = (match[2] || '').trim()
19
+ let lang = rawTag
20
+ let filename = explicitFile
21
+
22
+ // Handle ```ts:src/app.ts or ```typescript:app.ts
23
+ if (rawTag.includes(':')) {
24
+ const parts = rawTag.split(':')
25
+ lang = parts[0]
26
+ filename = parts.slice(1).join(':')
27
+ } else if (rawTag.includes('/') || rawTag.includes('.')) {
28
+ filename = rawTag
29
+ }
30
+
31
+ blocks.push({
32
+ lang: lang || 'text',
33
+ filename: filename || null,
34
+ code: match[3],
35
+ fullBlock: match[0],
36
+ })
37
+ }
38
+ return blocks
39
+ }
40
+
41
+ /**
42
+ * Builds composite block directives highlighting complementary modules across candidates.
43
+ */
44
+ export function buildCompositeBlockDirectives(candidates = []) {
45
+ const fileMap = new Map()
46
+
47
+ for (const c of candidates) {
48
+ if (!c.ok || !c.text) continue
49
+ const blocks = extractCodeBlocks(c.text)
50
+ for (const b of blocks) {
51
+ if (b.filename) {
52
+ if (!fileMap.has(b.filename)) {
53
+ fileMap.set(b.filename, [])
54
+ }
55
+ fileMap.get(b.filename).push(c.label || `Candidate ${c.index}`)
56
+ }
57
+ }
58
+ }
59
+
60
+ let fileOverview = ''
61
+ if (fileMap.size > 0) {
62
+ fileOverview = '\nDetected modular file contributions across candidates:\n' +
63
+ Array.from(fileMap.entries())
64
+ .map(([file, contributors]) => `- \`${file}\`: available in ${contributors.join(', ')}`)
65
+ .join('\n') + '\n'
66
+ }
67
+
68
+ return `
69
+ === COMPOSITE HYBRID SYNTHESIS DIRECTIVE ===
70
+ You must synthesize a unified, best-of-all-worlds composite solution from all candidates.
71
+ Do NOT simply copy one candidate. Instead:
72
+ 1. Extract the cleanest and most efficient core logic/algorithm (e.g. from the candidate with the highest reasoning score).
73
+ 2. Incorporate defensive error handling, input validation, and edge-case guards.
74
+ 3. Retain complete type annotations, utility helpers, and test coverage provided by any candidate.
75
+ 4. Ensure all merged modules, functions, and files fit together into a cohesive, non-conflicting solution.
76
+ ${fileOverview}============================================
77
+ `
78
+ }
79
+
80
+ /**
81
+ * Parses judge output into structured scores and candidate choice.
82
+ */
83
+ export function parseJudgeEvaluation(rawText = '', candidateCount = 2) {
84
+ const scores = {}
85
+ let bestCandidate = 1
86
+ let reasoning = ''
87
+
88
+ const scoreMatches = rawText.matchAll(/Candidate\s*(\d+)[:\s]+(?:Score\s*[:\s]*)?([0-9]+(?:\.[0-9]+)?)/gi)
89
+ for (const m of scoreMatches) {
90
+ const idx = parseInt(m[1], 10)
91
+ const val = parseFloat(m[2])
92
+ if (idx >= 1 && idx <= candidateCount && !Number.isNaN(val)) {
93
+ scores[idx] = Math.min(10, Math.max(0, val))
94
+ }
95
+ }
96
+
97
+ const bestMatch = rawText.match(/BEST_CANDIDATE\s*[:\s]+(?:Candidate\s*)?(\d+)/i)
98
+ if (bestMatch) {
99
+ const parsedBest = parseInt(bestMatch[1], 10)
100
+ if (parsedBest >= 1 && parsedBest <= candidateCount) {
101
+ bestCandidate = parsedBest
102
+ }
103
+ } else {
104
+ // Find candidate with max parsed score
105
+ let maxScore = -1
106
+ for (const [idxStr, s] of Object.entries(scores)) {
107
+ if (s > maxScore) {
108
+ maxScore = s
109
+ bestCandidate = parseInt(idxStr, 10)
110
+ }
111
+ }
112
+ }
113
+
114
+ const reasonMatch = rawText.match(/CONSENSUS_REASONING\s*[:\s]+([^\n\r]+)/i)
115
+ if (reasonMatch) {
116
+ reasoning = reasonMatch[1].trim()
117
+ } else {
118
+ reasoning = rawText.slice(0, 180).replace(/\n/g, ' ').trim()
119
+ }
120
+
121
+ return { scores, bestCandidate, reasoning }
122
+ }
123
+
124
+ /**
125
+ * Executes a panel of judges to reach consensus on candidate solutions.
126
+ */
127
+ export async function executeMultiJudgePanel({
128
+ judges = [],
129
+ candidates = [],
130
+ userPrompt = '',
131
+ callLlm,
132
+ strategy = 'majority', // 'majority' | 'highest_score' | 'unanimous'
133
+ timeoutMs = 45000,
134
+ signal,
135
+ onProgress,
136
+ }) {
137
+ const activeCandidates = (candidates || []).filter((c) => c && c.ok)
138
+ if (activeCandidates.length === 0) {
139
+ return {
140
+ consensus: false,
141
+ winningCandidateIndex: 1,
142
+ strategy,
143
+ reason: 'no_active_candidates',
144
+ }
145
+ }
146
+
147
+ if (activeCandidates.length === 1) {
148
+ return {
149
+ consensus: true,
150
+ winningCandidateIndex: activeCandidates[0].index,
151
+ winningCandidateLabel: activeCandidates[0].label,
152
+ strategy,
153
+ reason: 'single_active_candidate',
154
+ }
155
+ }
156
+
157
+ const judgeSlots = Array.isArray(judges) && judges.length > 0
158
+ ? judges
159
+ : [{ provider: 'default', model: 'aggregator-judge', label: 'Primary Judge' }]
160
+
161
+ const candidateSummary = activeCandidates.map((c) => {
162
+ return `### Candidate ${c.index} (${c.label}):\n${(c.text || '').slice(0, 4000)}`
163
+ }).join('\n\n---\n\n')
164
+
165
+ const judgePrompt = `You are an impartial Expert Code & Logic Judge evaluating multiple AI candidate solutions.
166
+ User Prompt:
167
+ ${userPrompt.slice(0, 1000)}
168
+
169
+ Candidate Solutions:
170
+ ${candidateSummary}
171
+
172
+ Evaluate the solutions on correctness, architecture, edge cases, and maintainability.
173
+ Rate each candidate on a scale of 1 to 10.
174
+ Pick the single best candidate.
175
+
176
+ Your response MUST follow this exact format:
177
+ SCORES:
178
+ ${activeCandidates.map((c) => `Candidate ${c.index}: [1-10]`).join('\n')}
179
+ BEST_CANDIDATE: [index number]
180
+ CONSENSUS_REASONING: [one concise sentence explaining your pick]
181
+ `
182
+
183
+ if (typeof onProgress === 'function') {
184
+ onProgress(`⚖️ *Convening Multi-Judge Panel (${judgeSlots.length} judges, strategy: ${strategy})...*\n`)
185
+ }
186
+
187
+ const judgePromises = judgeSlots.map(async (judge, jIdx) => {
188
+ const label = slotLabel(judge)
189
+ try {
190
+ const abortCtrl = new AbortController()
191
+ if (signal) {
192
+ signal.addEventListener('abort', () => abortCtrl.abort(), { once: true })
193
+ }
194
+ const timer = setTimeout(() => abortCtrl.abort(new Error('Judge evaluation timeout')), timeoutMs)
195
+ timer.unref?.()
196
+
197
+ const res = await callLlm({
198
+ provider: judge.provider,
199
+ model: judge.model,
200
+ messages: [
201
+ { role: 'system', content: 'You are an objective expert judge in an ensemble evaluation system.' },
202
+ { role: 'user', content: judgePrompt },
203
+ ],
204
+ temperature: 0.1,
205
+ maxTokens: 512,
206
+ signal: abortCtrl.signal,
207
+ })
208
+ clearTimeout(timer)
209
+
210
+ const text = typeof res === 'string' ? res : (res?.content || res?.text || '')
211
+ const evalResult = parseJudgeEvaluation(text, candidates.length)
212
+
213
+ return {
214
+ judgeIndex: jIdx + 1,
215
+ judgeLabel: label,
216
+ ok: true,
217
+ ...evalResult,
218
+ }
219
+ } catch (err) {
220
+ return {
221
+ judgeIndex: jIdx + 1,
222
+ judgeLabel: label,
223
+ ok: false,
224
+ bestCandidate: 1,
225
+ scores: {},
226
+ reasoning: err?.message || String(err),
227
+ }
228
+ }
229
+ })
230
+
231
+ const results = await Promise.all(judgePromises)
232
+ const validResults = results.filter((r) => r.ok)
233
+
234
+ // Tally votes and compute average scores
235
+ const votes = {}
236
+ const totalScores = {}
237
+ const scoreCounts = {}
238
+
239
+ for (const r of validResults) {
240
+ votes[r.bestCandidate] = (votes[r.bestCandidate] || 0) + 1
241
+ for (const [cIdx, s] of Object.entries(r.scores || {})) {
242
+ totalScores[cIdx] = (totalScores[cIdx] || 0) + s
243
+ scoreCounts[cIdx] = (scoreCounts[cIdx] || 0) + 1
244
+ }
245
+ }
246
+
247
+ const averageScores = {}
248
+ for (const c of activeCandidates) {
249
+ const cnt = scoreCounts[c.index] || 0
250
+ averageScores[c.index] = cnt > 0 ? Number((totalScores[c.index] / cnt).toFixed(2)) : 5.0
251
+ }
252
+
253
+ let winner = activeCandidates[0].index
254
+ let isUnanimous = false
255
+
256
+ if (strategy === 'highest_score') {
257
+ let highestAvg = -1
258
+ for (const c of activeCandidates) {
259
+ const avg = averageScores[c.index] || 0
260
+ if (avg > highestAvg) {
261
+ highestAvg = avg
262
+ winner = c.index
263
+ }
264
+ }
265
+ } else {
266
+ // majority or unanimous
267
+ let maxVotes = -1
268
+ for (const c of activeCandidates) {
269
+ const v = votes[c.index] || 0
270
+ if (v > maxVotes) {
271
+ maxVotes = v
272
+ winner = c.index
273
+ } else if (v === maxVotes && (averageScores[c.index] || 0) > (averageScores[winner] || 0)) {
274
+ winner = c.index
275
+ }
276
+ }
277
+ isUnanimous = (votes[winner] || 0) === validResults.length && validResults.length > 0
278
+ }
279
+
280
+ const winningCand = activeCandidates.find((c) => c.index === winner) || activeCandidates[0]
281
+
282
+ const consensusReport = `Multi-Judge Consensus (${validResults.length}/${judgeSlots.length} judges): ` +
283
+ `Winner is Candidate ${winningCand.index} (${winningCand.label}) with ${votes[winner] || 0} votes, ` +
284
+ `avg score ${averageScores[winner] || 0}/10. Strategy: ${strategy}.`
285
+
286
+ if (typeof onProgress === 'function') {
287
+ onProgress(`🏁 *${consensusReport}*\n`)
288
+ }
289
+
290
+ return {
291
+ consensus: true,
292
+ winningCandidateIndex: winningCand.index,
293
+ winningCandidateLabel: winningCand.label,
294
+ strategy,
295
+ isUnanimous,
296
+ votes,
297
+ averageScores,
298
+ judgesResults: results,
299
+ consensusReport,
300
+ }
301
+ }
@@ -343,6 +343,9 @@ ${ANTIPATTERNS_RUBRIC}
343
343
  ${LANGUAGE_MIRRORING_DIRECTIVE}
344
344
  ${referenceOutputs.some((r) => r.testResult) ? `\n${TEST_GATE_DIRECTIVE}\n` : ''}
345
345
  ${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
346
+ ${options.priorTurnBaseline ? `\n### 🔄 Prior Turn Context Baseline:\n${options.priorTurnBaseline}\n` : ''}
347
+ ${options.consensusReport ? `\n### ⚖️ Multi-Judge Consensus Advisory:\n${options.consensusReport}\n` : ''}
348
+ ${options.compositeMergeDirective ? `\n${options.compositeMergeDirective}\n` : ''}
346
349
  Instructions:
347
350
  Your response MUST be structured into three clear parts:
348
351
 
@@ -0,0 +1,124 @@
1
+ /**
2
+ * Automated Benchmark & Post-Mortem PR Reports
3
+ * Feature 9 (Automated Benchmark & Post-Mortem PR Reports)
4
+ */
5
+
6
+ export function generateMoABenchmarkReport({
7
+ runId = '',
8
+ timestamp = new Date().toISOString(),
9
+ preset = 'default',
10
+ prompt = '',
11
+ durationMs = 0,
12
+ usage = {},
13
+ costUsd = 0,
14
+ candidates = [],
15
+ winningIndex = 1,
16
+ winningLabel = '',
17
+ consensus = null,
18
+ testGate = null,
19
+ isComposite = false,
20
+ }) {
21
+ const durationSec = (durationMs / 1000).toFixed(2)
22
+ const totalCost = typeof costUsd === 'number' ? costUsd.toFixed(4) : '0.0000'
23
+ const inTokens = usage?.inputTokens || 0
24
+ const outTokens = usage?.outputTokens || 0
25
+ const totalTokens = usage?.totalTokens || (inTokens + outTokens)
26
+
27
+ const candidateRows = (candidates || []).map((c) => {
28
+ const idx = c.index || '?'
29
+ const label = c.label || `${c.slot?.provider}:${c.slot?.model}`
30
+ const role = c.role_persona || 'general'
31
+ const status = c.ok ? '✅ OK' : '❌ Failed'
32
+ const candTokens = c.usage?.totalTokens || (c.usage?.inputTokens || 0) + (c.usage?.outputTokens || 0)
33
+ const candCost = c.costUsd ? `$${c.costUsd.toFixed(4)}` : '$0.0000'
34
+ const score = consensus?.averageScores?.[c.index] !== undefined
35
+ ? `${consensus.averageScores[c.index]}/10`
36
+ : 'N/A'
37
+ const isWin = c.index === winningIndex ? ' 🏆' : ''
38
+ return `| ${idx}${isWin} | \`${label}\` | ${role} | ${status} | ${candTokens} | ${candCost} | ${score} |`
39
+ }).join('\n')
40
+
41
+ let testGateSection = '### 🧪 Automated Test Gate\n- **Status**: Not Configured / Skipped'
42
+ if (testGate && testGate.enabled) {
43
+ const tgStatus = testGate.passed ? '✅ PASSED' : '❌ FAILED'
44
+ testGateSection = `### 🧪 Automated Test Gate
45
+ - **Status**: ${tgStatus}
46
+ - **Test Command**: \`${testGate.command || 'N/A'}\`
47
+ - **Exit Code**: \`${testGate.exitCode ?? 0}\`
48
+ - **Output Snippet**:
49
+ \`\`\`
50
+ ${(testGate.output || 'No output recorded').slice(0, 500)}
51
+ \`\`\``
52
+ }
53
+
54
+ let consensusSection = '### ⚖️ Multi-Judge Consensus\n- **Status**: Standard single-aggregator synthesis'
55
+ if (consensus && consensus.consensus) {
56
+ consensusSection = `### ⚖️ Multi-Judge Consensus
57
+ - **Winning Candidate**: Candidate ${winningIndex} (${winningLabel})
58
+ - **Strategy**: \`${consensus.strategy || 'majority'}\`
59
+ - **Unanimous**: ${consensus.isUnanimous ? 'Yes ✅' : 'No (Majority)'}
60
+ - **Summary**: ${consensus.consensusReport || 'Consensus reached'}
61
+ `
62
+ }
63
+
64
+ const markdown = `# 🏆 Mixture of Agents (MoA) Benchmark & Post-Mortem Report
65
+
66
+ > **Run ID**: \`${runId}\`
67
+ > **Timestamp**: \`${timestamp}\`
68
+ > **Preset**: \`${preset}\`
69
+ > **Total Latency**: \`${durationSec}s\` | **Total Cost**: \`$${totalCost}\` | **Tokens**: \`${totalTokens}\` (In: ${inTokens}, Out: ${outTokens})
70
+
71
+ ---
72
+
73
+ ## 🎯 Task Prompt
74
+ \`\`\`text
75
+ ${prompt.slice(0, 500)}${prompt.length > 500 ? '...' : ''}
76
+ \`\`\`
77
+
78
+ ---
79
+
80
+ ## 📊 Candidate Models Benchmark Matrix
81
+ | # | Model / Provider | Role | Status | Tokens | Cost | Judge Score |
82
+ |---|------------------|------|--------|--------|------|-------------|
83
+ ${candidateRows || '| - | No candidate data | - | - | - | - | - |'}
84
+
85
+ ---
86
+
87
+ ${consensusSection}
88
+
89
+ ---
90
+
91
+ ${testGateSection}
92
+
93
+ ---
94
+
95
+ ### 🧬 Synthesis Mode
96
+ - **Mode**: ${isComposite ? 'Composite Hybrid AST/Block Merge (Multi-Candidate Synthesis)' : `Candidate Selection (Winning model: ${winningLabel || 'Aggregator'})`}
97
+ `
98
+
99
+ const json = {
100
+ runId,
101
+ timestamp,
102
+ preset,
103
+ prompt,
104
+ durationMs,
105
+ usage: { inputTokens: inTokens, outputTokens: outTokens, totalTokens },
106
+ costUsd,
107
+ candidates: (candidates || []).map((c) => ({
108
+ index: c.index,
109
+ label: c.label,
110
+ role: c.role_persona,
111
+ ok: c.ok,
112
+ costUsd: c.costUsd || 0,
113
+ usage: c.usage,
114
+ score: consensus?.averageScores?.[c.index] ?? null,
115
+ })),
116
+ winningIndex,
117
+ winningLabel,
118
+ consensus,
119
+ testGate,
120
+ isComposite,
121
+ }
122
+
123
+ return { markdown, json }
124
+ }