@goodandready/dsh-moa 0.2.25 → 0.2.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,124 @@
1
+ /**
2
+ * Automated Benchmark & Post-Mortem PR Reports
3
+ * Feature 9 (Automated Benchmark & Post-Mortem PR Reports)
4
+ */
5
+
6
+ export function generateMoABenchmarkReport({
7
+ runId = '',
8
+ timestamp = new Date().toISOString(),
9
+ preset = 'default',
10
+ prompt = '',
11
+ durationMs = 0,
12
+ usage = {},
13
+ costUsd = 0,
14
+ candidates = [],
15
+ winningIndex = 1,
16
+ winningLabel = '',
17
+ consensus = null,
18
+ testGate = null,
19
+ isComposite = false,
20
+ }) {
21
+ const durationSec = (durationMs / 1000).toFixed(2)
22
+ const totalCost = typeof costUsd === 'number' ? costUsd.toFixed(4) : '0.0000'
23
+ const inTokens = usage?.inputTokens || 0
24
+ const outTokens = usage?.outputTokens || 0
25
+ const totalTokens = usage?.totalTokens || (inTokens + outTokens)
26
+
27
+ const candidateRows = (candidates || []).map((c) => {
28
+ const idx = c.index || '?'
29
+ const label = c.label || `${c.slot?.provider}:${c.slot?.model}`
30
+ const role = c.role_persona || 'general'
31
+ const status = c.ok ? '✅ OK' : '❌ Failed'
32
+ const candTokens = c.usage?.totalTokens || (c.usage?.inputTokens || 0) + (c.usage?.outputTokens || 0)
33
+ const candCost = c.costUsd ? `$${c.costUsd.toFixed(4)}` : '$0.0000'
34
+ const score = consensus?.averageScores?.[c.index] !== undefined
35
+ ? `${consensus.averageScores[c.index]}/10`
36
+ : 'N/A'
37
+ const isWin = c.index === winningIndex ? ' 🏆' : ''
38
+ return `| ${idx}${isWin} | \`${label}\` | ${role} | ${status} | ${candTokens} | ${candCost} | ${score} |`
39
+ }).join('\n')
40
+
41
+ let testGateSection = '### 🧪 Automated Test Gate\n- **Status**: Not Configured / Skipped'
42
+ if (testGate && testGate.enabled) {
43
+ const tgStatus = testGate.passed ? '✅ PASSED' : '❌ FAILED'
44
+ testGateSection = `### 🧪 Automated Test Gate
45
+ - **Status**: ${tgStatus}
46
+ - **Test Command**: \`${testGate.command || 'N/A'}\`
47
+ - **Exit Code**: \`${testGate.exitCode ?? 0}\`
48
+ - **Output Snippet**:
49
+ \`\`\`
50
+ ${(testGate.output || 'No output recorded').slice(0, 500)}
51
+ \`\`\``
52
+ }
53
+
54
+ let consensusSection = '### ⚖️ Multi-Judge Consensus\n- **Status**: Standard single-aggregator synthesis'
55
+ if (consensus && consensus.consensus) {
56
+ consensusSection = `### ⚖️ Multi-Judge Consensus
57
+ - **Winning Candidate**: Candidate ${winningIndex} (${winningLabel})
58
+ - **Strategy**: \`${consensus.strategy || 'majority'}\`
59
+ - **Unanimous**: ${consensus.isUnanimous ? 'Yes ✅' : 'No (Majority)'}
60
+ - **Summary**: ${consensus.consensusReport || 'Consensus reached'}
61
+ `
62
+ }
63
+
64
+ const markdown = `# 🏆 Mixture of Agents (MoA) Benchmark & Post-Mortem Report
65
+
66
+ > **Run ID**: \`${runId}\`
67
+ > **Timestamp**: \`${timestamp}\`
68
+ > **Preset**: \`${preset}\`
69
+ > **Total Latency**: \`${durationSec}s\` | **Total Cost**: \`$${totalCost}\` | **Tokens**: \`${totalTokens}\` (In: ${inTokens}, Out: ${outTokens})
70
+
71
+ ---
72
+
73
+ ## 🎯 Task Prompt
74
+ \`\`\`text
75
+ ${prompt.slice(0, 500)}${prompt.length > 500 ? '...' : ''}
76
+ \`\`\`
77
+
78
+ ---
79
+
80
+ ## 📊 Candidate Models Benchmark Matrix
81
+ | # | Model / Provider | Role | Status | Tokens | Cost | Judge Score |
82
+ |---|------------------|------|--------|--------|------|-------------|
83
+ ${candidateRows || '| - | No candidate data | - | - | - | - | - |'}
84
+
85
+ ---
86
+
87
+ ${consensusSection}
88
+
89
+ ---
90
+
91
+ ${testGateSection}
92
+
93
+ ---
94
+
95
+ ### 🧬 Synthesis Mode
96
+ - **Mode**: ${isComposite ? 'Composite Hybrid AST/Block Merge (Multi-Candidate Synthesis)' : `Candidate Selection (Winning model: ${winningLabel || 'Aggregator'})`}
97
+ `
98
+
99
+ const json = {
100
+ runId,
101
+ timestamp,
102
+ preset,
103
+ prompt,
104
+ durationMs,
105
+ usage: { inputTokens: inTokens, outputTokens: outTokens, totalTokens },
106
+ costUsd,
107
+ candidates: (candidates || []).map((c) => ({
108
+ index: c.index,
109
+ label: c.label,
110
+ role: c.role_persona,
111
+ ok: c.ok,
112
+ costUsd: c.costUsd || 0,
113
+ usage: c.usage,
114
+ score: consensus?.averageScores?.[c.index] ?? null,
115
+ })),
116
+ winningIndex,
117
+ winningLabel,
118
+ consensus,
119
+ testGate,
120
+ isComposite,
121
+ }
122
+
123
+ return { markdown, json }
124
+ }
@@ -0,0 +1,102 @@
1
+ /**
2
+ * Smart Preset Router for Mixture of Agents (MoA).
3
+ * Supports heuristic keyword classification and optional JEV/LLM zero-shot routing.
4
+ */
5
+
6
+ const INTENT_RULES = [
7
+ { preset: 'security-audit', regex: /\b(security|vulnerability|exploit|cve|xss|injection|sanitize|threat|auth|permission|secret)\b/i },
8
+ { preset: 'bug-hunter', regex: /\b(bug|fix|error|crash|exception|regression|reproduce|broken|fail|issue)\b/i },
9
+ { preset: 'refactor-cleanup', regex: /\b(refactor|clean|simplify|ponytail|yagni|prune|dedup|modular|dead\s*code)\b/i },
10
+ { preset: 'code-review', regex: /\b(review|critique|audit|pr|pull\s*request|diff|inspect|check\s*code)\b/i },
11
+ { preset: 'frontend-ui', regex: /\b(frontend|ui|css|html|react|vue|tailwind|styling|styles?|stylesheet|component|button|layout|page|landing|modal)\b/i },
12
+ { preset: 'deep-architect', regex: /\b(architect|distributed|system|microservice|database|scalab|pipeline|infra|schema)\b/i },
13
+ { preset: 'math-logic', regex: /\b(math|algorithm|proof|matrix|calc|complexity|combinatorics|graph|tree|dynamic\s*programming)\b/i },
14
+ { preset: 'creative-brainstorm', regex: /\b(brainstorm|idea|concept|creative|alternative|options|feature\s*idea)\b/i },
15
+ { preset: 'fast-audit', regex: /\b(quick|fast|brief|summary|one-line|short|tldr)\b/i },
16
+ ]
17
+
18
+ /**
19
+ * Fast keyword-based intent classification.
20
+ */
21
+ export function classifyPromptIntent(prompt = '') {
22
+ if (!prompt || typeof prompt !== 'string') return null
23
+ const text = prompt.trim()
24
+ for (const rule of INTENT_RULES) {
25
+ if (rule.regex.test(text)) {
26
+ return rule.preset
27
+ }
28
+ }
29
+ return null
30
+ }
31
+
32
+ /**
33
+ * Resolves the optimal preset for a given user prompt.
34
+ */
35
+ export async function resolvePresetForPrompt({
36
+ prompt = '',
37
+ presets = [],
38
+ defaultPreset = 'default',
39
+ routingModel = null,
40
+ callLlm = null,
41
+ enabled = false,
42
+ timeoutMs = 2000,
43
+ }) {
44
+ const availableNames = (presets || []).map((p) => p.name).filter(Boolean)
45
+ const safeDefault = availableNames.includes(defaultPreset) ? defaultPreset : (availableNames[0] || 'default')
46
+
47
+ if (!enabled || !prompt || typeof prompt !== 'string') {
48
+ return { presetName: safeDefault, reason: 'default', isAutoRouted: false }
49
+ }
50
+
51
+ // 1. If LLM routing model (e.g. JEV) is configured and callable, run fast zero-shot classifier
52
+ if (routingModel?.provider && routingModel?.model && typeof callLlm === 'function') {
53
+ try {
54
+ const systemPrompt = `You are the MoA Preset Router. Given user prompt, classify which preset is best suited.
55
+ Available presets: ${availableNames.join(', ')}
56
+ Respond with ONLY the chosen preset name and nothing else.`
57
+
58
+ const abortCtrl = new AbortController()
59
+ const timer = setTimeout(() => abortCtrl.abort(), timeoutMs)
60
+ timer.unref?.()
61
+
62
+ const res = await callLlm({
63
+ provider: routingModel.provider,
64
+ model: routingModel.model,
65
+ messages: [
66
+ { role: 'system', content: systemPrompt },
67
+ { role: 'user', content: prompt.slice(0, 500) },
68
+ ],
69
+ temperature: 0.1,
70
+ maxTokens: 30,
71
+ signal: abortCtrl.signal,
72
+ })
73
+ clearTimeout(timer)
74
+
75
+ const rawChoice = typeof res === 'string' ? res : (res?.content || res?.text || '')
76
+ const match = availableNames.find((name) => new RegExp(`\\b${name}\\b`, 'i').test(rawChoice))
77
+ if (match) {
78
+ return {
79
+ presetName: match,
80
+ reason: 'llm_classifier',
81
+ routerModel: `${routingModel.provider}:${routingModel.model}`,
82
+ isAutoRouted: true,
83
+ }
84
+ }
85
+ } catch {
86
+ // Fallback silently to heuristic rules on timeout or network error
87
+ }
88
+ }
89
+
90
+ // 2. Keyword heuristic classifier
91
+ const heuristic = classifyPromptIntent(prompt)
92
+ if (heuristic && availableNames.includes(heuristic)) {
93
+ return {
94
+ presetName: heuristic,
95
+ reason: 'keyword_heuristic',
96
+ isAutoRouted: true,
97
+ }
98
+ }
99
+
100
+ // 3. Fallback to default
101
+ return { presetName: safeDefault, reason: 'default_fallback', isAutoRouted: false }
102
+ }
package/lib/moa-runner.js CHANGED
@@ -1,28 +1,26 @@
1
- import { bestEffort } from './best-effort.js'
2
- /**
3
- * DeepSeek Harness Mixture of Agents (MoA) — Runner Engine
4
- * Parallel fan-out, Consilium Round 2 peer critique, aggregator synthesis & streaming.
5
- */
6
-
7
1
  import path from 'node:path'
8
2
  import crypto from 'node:crypto'
3
+ import { bestEffort } from './best-effort.js'
9
4
  import { extractFileBlocks, collectProjectContext, formatProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
10
5
  import { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
11
- import { recordMoaRun, recordMoaRunAsync, candidatesForHistory } from './history.js'
6
+ import { recordMoaRun, candidatesForHistory } from './history.js'
12
7
  import { createPromotedPreview } from './live-canvas.js'
13
8
  import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
14
9
  import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
10
+ import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
11
+ import { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
12
+ import { executeMultiJudgePanel, buildCompositeBlockDirectives } from './moa-multi-judge.js'
13
+ import { applyBudgetGuardrails } from './moa-budget.js'
14
+ import { extractPriorTurnBaseline, pruneMultiTurnMessages } from './moa-context.js'
15
+ import { generateMoABenchmarkReport } from './moa-report.js'
15
16
 
16
17
  // Re-exports for consumers & backward compatibility
17
18
  export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
18
19
  export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
19
20
  export { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
20
-
21
- /**
22
- * Invokes LLM call with transient retry for recoverable network/rate-limit errors.
23
- */
24
- import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
25
- export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel }
21
+ export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
22
+ export { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
23
+ export { streamMoATurn } from './moa-stream.js'
26
24
 
27
25
  /**
28
26
  * Executes the full Mixture of Agents pipeline.
@@ -39,18 +37,18 @@ export async function runMoAPipeline({
39
37
  liveCanvas = null,
40
38
  onStreamDelta = null,
41
39
  signal = null,
40
+ testGateEnabled = null,
41
+ testCommand = null,
42
+ execFn = null,
42
43
  }) {
43
44
  if (signal?.aborted) return { error: new Error('Turn aborted') }
44
45
  const startTime = Date.now()
45
46
 
46
47
  // 1. Resolve configurations
47
48
  const referenceModels = Array.isArray(preset?.reference_models) && preset.reference_models.length > 0
48
- ? preset.reference_models
49
- : [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
50
-
49
+ ? preset.reference_models : [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
51
50
  const primaryJudge = preset?.aggregator?.provider && preset?.aggregator?.model
52
- ? preset.aggregator
53
- : { provider: 'codex', model: 'gpt-5.6-sol' }
51
+ ? preset.aggregator : { provider: 'codex', model: 'gpt-5.6-sol' }
54
52
 
55
53
  const fallbackJudges = Array.isArray(preset?.aggregator_fallbacks) ? preset.aggregator_fallbacks : []
56
54
  const judgesChain = [primaryJudge, ...fallbackJudges]
@@ -60,17 +58,44 @@ export async function runMoAPipeline({
60
58
  const maxTokens = typeof preset?.max_tokens === 'number' ? preset.max_tokens : 4096
61
59
  const judgeCriteria = preset?.judge_criteria || ''
62
60
  const isFastMode = referenceModels.length === 1 && !preset?.curator_synthesis
63
- const isCuratorSynthesis = Boolean(preset?.curator_synthesis)
64
- const isStreamAggregator = preset?.stream_aggregator !== false
65
- const isQuorumEnabled = Boolean(preset?.quorum_enabled)
66
- const gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
67
- const candidateRetries = 1
68
- const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
61
+ const isCuratorSynthesis = Boolean(preset?.curator_synthesis), isStreamAggregator = preset?.stream_aggregator !== false
62
+ const isQuorumEnabled = Boolean(preset?.quorum_enabled), gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
63
+ const candidateRetries = 1, refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
69
64
  const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
70
- const isBlindEvaluation = Boolean(preset?.blind_evaluation)
71
- const isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
72
- const allowCandidateOverride = Boolean(preset?.allow_candidate_override)
73
- const runId = crypto.randomUUID()
65
+ const isBlindEvaluation = Boolean(preset?.blind_evaluation), isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
66
+ const allowCandidateOverride = Boolean(preset?.allow_candidate_override), runId = crypto.randomUUID()
67
+
68
+ // Feature 6: Budget Guardrail check
69
+ const budgetCheck = applyBudgetGuardrails({
70
+ references: referenceModels,
71
+ enabled: preset?.budget_guard_enabled,
72
+ maxBudgetUsd: preset?.max_budget_usd,
73
+ action: preset?.budget_action || 'trim',
74
+ prices,
75
+ promptLength: userPrompt?.length || 1000,
76
+ })
77
+ if (!budgetCheck.allowed) {
78
+ return {
79
+ kind: 'failure',
80
+ content: `🛑 Budget Guardrail Abort: ${budgetCheck.reason}`,
81
+ aggregator: slotLabel(primaryJudge),
82
+ references: [],
83
+ presetName: preset?.name || 'default',
84
+ isRefinement: false,
85
+ isFastMode,
86
+ winningIndex: 0,
87
+ winnerModel: 'none',
88
+ promotedFiles: [],
89
+ usage: { totalTokens: 0, totalCostUsd: 0, candidates: [], aggregator: { totalTokens: 0, costUsd: 0 } },
90
+ skippedFiles: 0,
91
+ skippedList: [],
92
+ durationMs: Date.now() - startTime,
93
+ }
94
+ }
95
+ const effectiveReferences = budgetCheck.references
96
+ if (budgetCheck.action === 'trim' && typeof onProgress === 'function') {
97
+ onProgress(`✂️ *${budgetCheck.reason}*\n`)
98
+ }
74
99
 
75
100
  // 2. Collect project context for refinement tasks
76
101
  const collectedCtx = await collectProjectContext(cwd, 16000)
@@ -91,23 +116,28 @@ export async function runMoAPipeline({
91
116
  })
92
117
  }
93
118
 
94
- // 3. Build prompts & evaluate broad questionnaire needs
119
+ // 3. Build prompts & Feature 8 multi-turn context
120
+ const priorTurnBaseline = preset?.multi_turn_enabled !== false ? extractPriorTurnBaseline(messages) : null
121
+ const prunedHistory = preset?.multi_turn_enabled !== false ? pruneMultiTurnMessages(messages) : messages
122
+
95
123
  const candidateSystemPrompt = projectContext
96
124
  ? `${REFERENCE_SYSTEM_PROMPT}\n\n### Current Project Files & Context:\n${projectContext}`
97
125
  : REFERENCE_SYSTEM_PROMPT
98
126
 
99
127
  const askClarifyingQuestions = preset?.ask_clarifying_questions !== false
100
128
  const needsQuestions = askClarifyingQuestions && isBroadPromptRequiringQuestions(userPrompt, messages)
129
+ const enrichedMessages = [...prunedHistory, { role: 'user', content: userPrompt }]
101
130
 
102
- const enrichedMessages = [...messages, { role: 'user', content: userPrompt }]
103
-
104
- // 4. Parallel fan-out to candidate models with quorum & transient retry
131
+ // 4. Parallel fan-out to candidate models with temperature gradient & local fallback
105
132
  const referenceOutputs = await runReferencesParallel(
106
- referenceModels,
133
+ effectiveReferences,
107
134
  enrichedMessages,
108
135
  {
109
136
  systemPrompt: candidateSystemPrompt,
110
137
  temperature: refTemp,
138
+ temperature_gradient_enabled: Boolean(preset?.temperature_gradient_enabled),
139
+ local_fallback_enabled: Boolean(preset?.local_fallback_enabled),
140
+ local_fallback_models: preset?.local_fallback_models || [],
111
141
  maxTokens,
112
142
  prices,
113
143
  timeoutSec: refTimeoutSec,
@@ -159,10 +189,20 @@ export async function runMoAPipeline({
159
189
  const files = extractFileBlocks(ref.text)
160
190
  ref.files = files
161
191
  if (files.length > 0) {
192
+ ref.syntaxWarning = (verifyFileSyntax(files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
162
193
  await writeCandidateWorkspace(cwd, i + 1, files)
163
194
  }
164
195
  }
165
196
 
197
+ // 5b. Test Execution Gate
198
+ await executeTestGateForCandidates({
199
+ cwd,
200
+ referenceOutputs,
201
+ preset,
202
+ options: { testGateEnabled, testCommand, execFn },
203
+ onProgress,
204
+ })
205
+
166
206
  // 6. Questionnaire synthesis branch
167
207
  if (needsQuestions) {
168
208
  if (typeof onProgress === 'function') {
@@ -188,11 +228,7 @@ export async function runMoAPipeline({
188
228
  qUsage = estimateTokenCost(primaryJudge, qFallbackUsage, prices)
189
229
  } catch (err) {
190
230
  console.warn('[dsh-moa] Questionnaire synthesis failed, proceeding with fallback questions:', err)
191
- questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n` +
192
- '1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n' +
193
- '2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n' +
194
- '3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n' +
195
- '*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*'
231
+ questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*`
196
232
  }
197
233
 
198
234
  await cleanMoaWorkspaces(cwd)
@@ -344,12 +380,41 @@ export async function runMoAPipeline({
344
380
  }
345
381
  })
346
382
  await Promise.allSettled(r2Promises)
383
+ await executeTestGateForCandidates({
384
+ cwd,
385
+ referenceOutputs: successfulRefs,
386
+ preset,
387
+ options: { testGateEnabled, testCommand, execFn },
388
+ onProgress,
389
+ })
347
390
  }
348
391
 
392
+ // 7c. Feature 2: Multi-Judge Panel & Consensus Voting
393
+ let multiJudgeResult = null
394
+ if (preset?.multi_judge_enabled && successfulRefs.length > 1) {
395
+ multiJudgeResult = await executeMultiJudgePanel({
396
+ judges: (preset.judge_models && preset.judge_models.length > 0) ? preset.judge_models : [primaryJudge],
397
+ candidates: successfulRefs,
398
+ userPrompt,
399
+ callLlm,
400
+ strategy: preset.judge_voting_strategy || 'majority',
401
+ signal,
402
+ onProgress,
403
+ })
404
+ }
405
+
406
+ // 7d. Feature 3: Composite Block Merge Directive
407
+ const compositeMergeDirective = (preset?.composite_merge_enabled && successfulRefs.length > 1)
408
+ ? buildCompositeBlockDirectives(successfulRefs)
409
+ : null
410
+
349
411
  // 8. Synthesis phase via primary judge or fallback chain
350
412
  const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
351
413
  curatorSynthesis: isCuratorSynthesis,
352
414
  blindEvaluation: isBlindEvaluation,
415
+ priorTurnBaseline,
416
+ compositeMergeDirective,
417
+ consensusReport: multiJudgeResult?.consensusReport,
353
418
  })
354
419
  let synthesizedText = ''
355
420
  let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
@@ -411,7 +476,10 @@ export async function runMoAPipeline({
411
476
  }
412
477
 
413
478
  // 9. Evaluate winner & promote files
414
- const winningIndex = parseWinnerIndex(synthesizedText, 1, referenceOutputs.length)
479
+ const winningIndex = (multiJudgeResult?.consensus && multiJudgeResult?.winningCandidateIndex)
480
+ ? multiJudgeResult.winningCandidateIndex
481
+ : parseWinnerIndex(synthesizedText, 1, referenceOutputs.length)
482
+
415
483
  const recommendedAssembler = isCuratorSynthesis
416
484
  ? parseRecommendedAssembler(synthesizedText, winningIndex, referenceOutputs.length)
417
485
  : null
@@ -434,13 +502,32 @@ export async function runMoAPipeline({
434
502
  }
435
503
 
436
504
  const livePreview = await createPromotedPreview(liveCanvas, cwd, promotedFiles, referenceOutputs)
437
-
438
505
  const durationMs = Date.now() - startTime
439
506
  const usageSummary = summarizeMoAUsage(referenceOutputs, aggUsage)
440
507
  const winningRef = referenceOutputs[winningIndex - 1]
441
508
  const winnerModel = winningRef?.label || slotLabel(referenceModels[0])
442
509
  const finalAggLabel = slotLabel(chosenJudge)
443
510
 
511
+ // Feature 9: Benchmark & Post-Mortem PR report
512
+ let benchmarkReport = null
513
+ if (preset?.report_generation_enabled) {
514
+ benchmarkReport = generateMoABenchmarkReport({
515
+ runId,
516
+ timestamp: new Date().toISOString(),
517
+ preset: preset?.name || 'default',
518
+ prompt: userPrompt,
519
+ durationMs,
520
+ usage: usageSummary,
521
+ costUsd: usageSummary.totalCostUsd,
522
+ candidates: referenceOutputs,
523
+ winningIndex,
524
+ winningLabel: winnerModel,
525
+ consensus: multiJudgeResult,
526
+ testGate: referenceOutputs.find((r) => r.testResult)?.testResult || null,
527
+ isComposite: Boolean(preset?.composite_merge_enabled),
528
+ })
529
+ }
530
+
444
531
  try {
445
532
  recordMoaRun({
446
533
  id: runId,
@@ -452,6 +539,8 @@ export async function runMoAPipeline({
452
539
  winnerIndex: winningIndex,
453
540
  winnerModel,
454
541
  promotedFiles,
542
+ benchmarkReport,
543
+ consensus: multiJudgeResult,
455
544
  totalTokens: usageSummary.totalTokens,
456
545
  totalCostUsd: usageSummary.totalCostUsd,
457
546
  durationMs,
@@ -475,6 +564,8 @@ export async function runMoAPipeline({
475
564
  promotedFiles,
476
565
  runId,
477
566
  allowCandidateOverride,
567
+ benchmarkReport,
568
+ consensus: multiJudgeResult,
478
569
  ...(livePreview ? { liveCanvas: livePreview } : {}),
479
570
  usage: usageSummary,
480
571
  skippedFiles: collectedCtx?.skippedFiles || 0,
@@ -482,115 +573,3 @@ export async function runMoAPipeline({
482
573
  durationMs,
483
574
  }
484
575
  }
485
-
486
- /**
487
- * Streams a full MoA turn into chat with live aggregator tokens and progress feedback.
488
- */
489
- export async function* streamMoATurn({ targetPreset, userPrompt, messages, callLlm, cwd, prices, historyFilePath, liveCanvas }, options = {}) {
490
- const signal = options?.signal
491
- if (signal?.aborted) return
492
-
493
- yield { type: 'block-start', index: 0, blockType: 'text' }
494
- yield { type: 'text-delta', index: 0, text: '🧠 *Mixture of Agents started...*\n\n' }
495
-
496
- // Async push-queue for zero-latency live delta streaming
497
- const queue = []
498
- let notify = null
499
-
500
- const pushUpdate = (text) => {
501
- queue.push({ type: 'text-delta', index: 0, text })
502
- if (notify) {
503
- notify()
504
- notify = null
505
- }
506
- }
507
-
508
- let done = false
509
- const pipelinePromise = runMoAPipeline({
510
- userPrompt,
511
- messages,
512
- preset: targetPreset,
513
- callLlm,
514
- cwd,
515
- onProgress: pushUpdate,
516
- onStreamDelta: (delta) => {
517
- pushUpdate(delta)
518
- },
519
- prices,
520
- historyFilePath,
521
- liveCanvas,
522
- signal,
523
- })
524
- .catch((err) => ({ error: err }))
525
- .finally(() => {
526
- done = true
527
- if (notify) {
528
- notify()
529
- notify = null
530
- }
531
- })
532
-
533
- const startTime = Date.now()
534
- let lastYieldTime = Date.now()
535
-
536
- try {
537
- while (!done || queue.length > 0) {
538
- if (signal?.aborted) {
539
- await cleanMoaWorkspaces(cwd)
540
- return
541
- }
542
-
543
- while (queue.length > 0) {
544
- const item = queue.shift()
545
- yield item
546
- lastYieldTime = Date.now()
547
- }
548
-
549
- if (done) break
550
-
551
- // Wait for next push item or max 2.5s heartbeat
552
- await Promise.race([
553
- new Promise((resolve) => { notify = resolve }),
554
- new Promise((resolve) => { const t = setTimeout(resolve, 2500); t.unref?.(); }),
555
- ])
556
-
557
- const elapsedSec = Math.floor((Date.now() - startTime) / 1000)
558
- if (!done && Date.now() - lastYieldTime >= 3000) {
559
- yield { type: 'text-delta', index: 0, text: `⏳ *[${elapsedSec}s] Still processing...*\n` }
560
- lastYieldTime = Date.now()
561
- }
562
- }
563
-
564
- const result = await pipelinePromise
565
- if (signal?.aborted) {
566
- await cleanMoaWorkspaces(cwd)
567
- return
568
- }
569
-
570
- if (result?.error) {
571
- const errText = '\n\n⚠️ **Mixture of Agents error**: ' + (result.error?.message || String(result.error))
572
- yield { type: 'text-delta', index: 0, text: errText }
573
- yield { type: 'block-end', index: 0, block: { type: 'text', text: errText } }
574
- yield { type: 'finish', reason: { kind: 'stop' } }
575
- return
576
- }
577
-
578
- const formatted = formatMoAResponse({
579
- moaResult: result,
580
- presetName: targetPreset?.name || 'default',
581
- })
582
-
583
- yield { type: 'text-delta', index: 0, text: '\n---\n\n' + formatted }
584
- yield { type: 'block-end', index: 0, block: { type: 'text', text: formatted } }
585
- yield {
586
- type: 'usage',
587
- usage: {
588
- inputTokens: result?.usage?.totalTokens || 0,
589
- outputTokens: Math.round((formatted.length || 0) / 4),
590
- },
591
- }
592
- yield { type: 'finish', reason: { kind: 'stop' } }
593
- } finally {
594
- // cleanup
595
- }
596
- }