@goodandready/dsh-moa 0.2.25 → 0.2.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/README.ru.md +30 -0
- package/lib/client.js +248 -0
- package/lib/history.js +11 -0
- package/lib/index.js +41 -0
- package/lib/moa-budget.js +97 -0
- package/lib/moa-candidates.js +90 -22
- package/lib/moa-context.js +70 -0
- package/lib/moa-multi-judge.js +301 -0
- package/lib/moa-prompts.js +25 -6
- package/lib/moa-report.js +124 -0
- package/lib/moa-router.js +102 -0
- package/lib/moa-runner.js +131 -152
- package/lib/moa-stream.js +117 -0
- package/lib/moa-test-gate.js +236 -0
- package/lib/routes.js +58 -0
- package/package.json +1 -1
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Automated Benchmark & Post-Mortem PR Reports
|
|
3
|
+
* Feature 9 (Automated Benchmark & Post-Mortem PR Reports)
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
export function generateMoABenchmarkReport({
|
|
7
|
+
runId = '',
|
|
8
|
+
timestamp = new Date().toISOString(),
|
|
9
|
+
preset = 'default',
|
|
10
|
+
prompt = '',
|
|
11
|
+
durationMs = 0,
|
|
12
|
+
usage = {},
|
|
13
|
+
costUsd = 0,
|
|
14
|
+
candidates = [],
|
|
15
|
+
winningIndex = 1,
|
|
16
|
+
winningLabel = '',
|
|
17
|
+
consensus = null,
|
|
18
|
+
testGate = null,
|
|
19
|
+
isComposite = false,
|
|
20
|
+
}) {
|
|
21
|
+
const durationSec = (durationMs / 1000).toFixed(2)
|
|
22
|
+
const totalCost = typeof costUsd === 'number' ? costUsd.toFixed(4) : '0.0000'
|
|
23
|
+
const inTokens = usage?.inputTokens || 0
|
|
24
|
+
const outTokens = usage?.outputTokens || 0
|
|
25
|
+
const totalTokens = usage?.totalTokens || (inTokens + outTokens)
|
|
26
|
+
|
|
27
|
+
const candidateRows = (candidates || []).map((c) => {
|
|
28
|
+
const idx = c.index || '?'
|
|
29
|
+
const label = c.label || `${c.slot?.provider}:${c.slot?.model}`
|
|
30
|
+
const role = c.role_persona || 'general'
|
|
31
|
+
const status = c.ok ? '✅ OK' : '❌ Failed'
|
|
32
|
+
const candTokens = c.usage?.totalTokens || (c.usage?.inputTokens || 0) + (c.usage?.outputTokens || 0)
|
|
33
|
+
const candCost = c.costUsd ? `$${c.costUsd.toFixed(4)}` : '$0.0000'
|
|
34
|
+
const score = consensus?.averageScores?.[c.index] !== undefined
|
|
35
|
+
? `${consensus.averageScores[c.index]}/10`
|
|
36
|
+
: 'N/A'
|
|
37
|
+
const isWin = c.index === winningIndex ? ' 🏆' : ''
|
|
38
|
+
return `| ${idx}${isWin} | \`${label}\` | ${role} | ${status} | ${candTokens} | ${candCost} | ${score} |`
|
|
39
|
+
}).join('\n')
|
|
40
|
+
|
|
41
|
+
let testGateSection = '### 🧪 Automated Test Gate\n- **Status**: Not Configured / Skipped'
|
|
42
|
+
if (testGate && testGate.enabled) {
|
|
43
|
+
const tgStatus = testGate.passed ? '✅ PASSED' : '❌ FAILED'
|
|
44
|
+
testGateSection = `### 🧪 Automated Test Gate
|
|
45
|
+
- **Status**: ${tgStatus}
|
|
46
|
+
- **Test Command**: \`${testGate.command || 'N/A'}\`
|
|
47
|
+
- **Exit Code**: \`${testGate.exitCode ?? 0}\`
|
|
48
|
+
- **Output Snippet**:
|
|
49
|
+
\`\`\`
|
|
50
|
+
${(testGate.output || 'No output recorded').slice(0, 500)}
|
|
51
|
+
\`\`\``
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
let consensusSection = '### ⚖️ Multi-Judge Consensus\n- **Status**: Standard single-aggregator synthesis'
|
|
55
|
+
if (consensus && consensus.consensus) {
|
|
56
|
+
consensusSection = `### ⚖️ Multi-Judge Consensus
|
|
57
|
+
- **Winning Candidate**: Candidate ${winningIndex} (${winningLabel})
|
|
58
|
+
- **Strategy**: \`${consensus.strategy || 'majority'}\`
|
|
59
|
+
- **Unanimous**: ${consensus.isUnanimous ? 'Yes ✅' : 'No (Majority)'}
|
|
60
|
+
- **Summary**: ${consensus.consensusReport || 'Consensus reached'}
|
|
61
|
+
`
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const markdown = `# 🏆 Mixture of Agents (MoA) Benchmark & Post-Mortem Report
|
|
65
|
+
|
|
66
|
+
> **Run ID**: \`${runId}\`
|
|
67
|
+
> **Timestamp**: \`${timestamp}\`
|
|
68
|
+
> **Preset**: \`${preset}\`
|
|
69
|
+
> **Total Latency**: \`${durationSec}s\` | **Total Cost**: \`$${totalCost}\` | **Tokens**: \`${totalTokens}\` (In: ${inTokens}, Out: ${outTokens})
|
|
70
|
+
|
|
71
|
+
---
|
|
72
|
+
|
|
73
|
+
## 🎯 Task Prompt
|
|
74
|
+
\`\`\`text
|
|
75
|
+
${prompt.slice(0, 500)}${prompt.length > 500 ? '...' : ''}
|
|
76
|
+
\`\`\`
|
|
77
|
+
|
|
78
|
+
---
|
|
79
|
+
|
|
80
|
+
## 📊 Candidate Models Benchmark Matrix
|
|
81
|
+
| # | Model / Provider | Role | Status | Tokens | Cost | Judge Score |
|
|
82
|
+
|---|------------------|------|--------|--------|------|-------------|
|
|
83
|
+
${candidateRows || '| - | No candidate data | - | - | - | - | - |'}
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
${consensusSection}
|
|
88
|
+
|
|
89
|
+
---
|
|
90
|
+
|
|
91
|
+
${testGateSection}
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
### 🧬 Synthesis Mode
|
|
96
|
+
- **Mode**: ${isComposite ? 'Composite Hybrid AST/Block Merge (Multi-Candidate Synthesis)' : `Candidate Selection (Winning model: ${winningLabel || 'Aggregator'})`}
|
|
97
|
+
`
|
|
98
|
+
|
|
99
|
+
const json = {
|
|
100
|
+
runId,
|
|
101
|
+
timestamp,
|
|
102
|
+
preset,
|
|
103
|
+
prompt,
|
|
104
|
+
durationMs,
|
|
105
|
+
usage: { inputTokens: inTokens, outputTokens: outTokens, totalTokens },
|
|
106
|
+
costUsd,
|
|
107
|
+
candidates: (candidates || []).map((c) => ({
|
|
108
|
+
index: c.index,
|
|
109
|
+
label: c.label,
|
|
110
|
+
role: c.role_persona,
|
|
111
|
+
ok: c.ok,
|
|
112
|
+
costUsd: c.costUsd || 0,
|
|
113
|
+
usage: c.usage,
|
|
114
|
+
score: consensus?.averageScores?.[c.index] ?? null,
|
|
115
|
+
})),
|
|
116
|
+
winningIndex,
|
|
117
|
+
winningLabel,
|
|
118
|
+
consensus,
|
|
119
|
+
testGate,
|
|
120
|
+
isComposite,
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
return { markdown, json }
|
|
124
|
+
}
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Smart Preset Router for Mixture of Agents (MoA).
|
|
3
|
+
* Supports heuristic keyword classification and optional JEV/LLM zero-shot routing.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
const INTENT_RULES = [
|
|
7
|
+
{ preset: 'security-audit', regex: /\b(security|vulnerability|exploit|cve|xss|injection|sanitize|threat|auth|permission|secret)\b/i },
|
|
8
|
+
{ preset: 'bug-hunter', regex: /\b(bug|fix|error|crash|exception|regression|reproduce|broken|fail|issue)\b/i },
|
|
9
|
+
{ preset: 'refactor-cleanup', regex: /\b(refactor|clean|simplify|ponytail|yagni|prune|dedup|modular|dead\s*code)\b/i },
|
|
10
|
+
{ preset: 'code-review', regex: /\b(review|critique|audit|pr|pull\s*request|diff|inspect|check\s*code)\b/i },
|
|
11
|
+
{ preset: 'frontend-ui', regex: /\b(frontend|ui|css|html|react|vue|tailwind|styling|styles?|stylesheet|component|button|layout|page|landing|modal)\b/i },
|
|
12
|
+
{ preset: 'deep-architect', regex: /\b(architect|distributed|system|microservice|database|scalab|pipeline|infra|schema)\b/i },
|
|
13
|
+
{ preset: 'math-logic', regex: /\b(math|algorithm|proof|matrix|calc|complexity|combinatorics|graph|tree|dynamic\s*programming)\b/i },
|
|
14
|
+
{ preset: 'creative-brainstorm', regex: /\b(brainstorm|idea|concept|creative|alternative|options|feature\s*idea)\b/i },
|
|
15
|
+
{ preset: 'fast-audit', regex: /\b(quick|fast|brief|summary|one-line|short|tldr)\b/i },
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Fast keyword-based intent classification.
|
|
20
|
+
*/
|
|
21
|
+
export function classifyPromptIntent(prompt = '') {
|
|
22
|
+
if (!prompt || typeof prompt !== 'string') return null
|
|
23
|
+
const text = prompt.trim()
|
|
24
|
+
for (const rule of INTENT_RULES) {
|
|
25
|
+
if (rule.regex.test(text)) {
|
|
26
|
+
return rule.preset
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
return null
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Resolves the optimal preset for a given user prompt.
|
|
34
|
+
*/
|
|
35
|
+
export async function resolvePresetForPrompt({
|
|
36
|
+
prompt = '',
|
|
37
|
+
presets = [],
|
|
38
|
+
defaultPreset = 'default',
|
|
39
|
+
routingModel = null,
|
|
40
|
+
callLlm = null,
|
|
41
|
+
enabled = false,
|
|
42
|
+
timeoutMs = 2000,
|
|
43
|
+
}) {
|
|
44
|
+
const availableNames = (presets || []).map((p) => p.name).filter(Boolean)
|
|
45
|
+
const safeDefault = availableNames.includes(defaultPreset) ? defaultPreset : (availableNames[0] || 'default')
|
|
46
|
+
|
|
47
|
+
if (!enabled || !prompt || typeof prompt !== 'string') {
|
|
48
|
+
return { presetName: safeDefault, reason: 'default', isAutoRouted: false }
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// 1. If LLM routing model (e.g. JEV) is configured and callable, run fast zero-shot classifier
|
|
52
|
+
if (routingModel?.provider && routingModel?.model && typeof callLlm === 'function') {
|
|
53
|
+
try {
|
|
54
|
+
const systemPrompt = `You are the MoA Preset Router. Given user prompt, classify which preset is best suited.
|
|
55
|
+
Available presets: ${availableNames.join(', ')}
|
|
56
|
+
Respond with ONLY the chosen preset name and nothing else.`
|
|
57
|
+
|
|
58
|
+
const abortCtrl = new AbortController()
|
|
59
|
+
const timer = setTimeout(() => abortCtrl.abort(), timeoutMs)
|
|
60
|
+
timer.unref?.()
|
|
61
|
+
|
|
62
|
+
const res = await callLlm({
|
|
63
|
+
provider: routingModel.provider,
|
|
64
|
+
model: routingModel.model,
|
|
65
|
+
messages: [
|
|
66
|
+
{ role: 'system', content: systemPrompt },
|
|
67
|
+
{ role: 'user', content: prompt.slice(0, 500) },
|
|
68
|
+
],
|
|
69
|
+
temperature: 0.1,
|
|
70
|
+
maxTokens: 30,
|
|
71
|
+
signal: abortCtrl.signal,
|
|
72
|
+
})
|
|
73
|
+
clearTimeout(timer)
|
|
74
|
+
|
|
75
|
+
const rawChoice = typeof res === 'string' ? res : (res?.content || res?.text || '')
|
|
76
|
+
const match = availableNames.find((name) => new RegExp(`\\b${name}\\b`, 'i').test(rawChoice))
|
|
77
|
+
if (match) {
|
|
78
|
+
return {
|
|
79
|
+
presetName: match,
|
|
80
|
+
reason: 'llm_classifier',
|
|
81
|
+
routerModel: `${routingModel.provider}:${routingModel.model}`,
|
|
82
|
+
isAutoRouted: true,
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
} catch {
|
|
86
|
+
// Fallback silently to heuristic rules on timeout or network error
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// 2. Keyword heuristic classifier
|
|
91
|
+
const heuristic = classifyPromptIntent(prompt)
|
|
92
|
+
if (heuristic && availableNames.includes(heuristic)) {
|
|
93
|
+
return {
|
|
94
|
+
presetName: heuristic,
|
|
95
|
+
reason: 'keyword_heuristic',
|
|
96
|
+
isAutoRouted: true,
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// 3. Fallback to default
|
|
101
|
+
return { presetName: safeDefault, reason: 'default_fallback', isAutoRouted: false }
|
|
102
|
+
}
|
package/lib/moa-runner.js
CHANGED
|
@@ -1,28 +1,26 @@
|
|
|
1
|
-
import { bestEffort } from './best-effort.js'
|
|
2
|
-
/**
|
|
3
|
-
* DeepSeek Harness Mixture of Agents (MoA) — Runner Engine
|
|
4
|
-
* Parallel fan-out, Consilium Round 2 peer critique, aggregator synthesis & streaming.
|
|
5
|
-
*/
|
|
6
|
-
|
|
7
1
|
import path from 'node:path'
|
|
8
2
|
import crypto from 'node:crypto'
|
|
3
|
+
import { bestEffort } from './best-effort.js'
|
|
9
4
|
import { extractFileBlocks, collectProjectContext, formatProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
|
|
10
5
|
import { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
|
|
11
|
-
import { recordMoaRun,
|
|
6
|
+
import { recordMoaRun, candidatesForHistory } from './history.js'
|
|
12
7
|
import { createPromotedPreview } from './live-canvas.js'
|
|
13
8
|
import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
14
9
|
import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
10
|
+
import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
11
|
+
import { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
|
12
|
+
import { executeMultiJudgePanel, buildCompositeBlockDirectives } from './moa-multi-judge.js'
|
|
13
|
+
import { applyBudgetGuardrails } from './moa-budget.js'
|
|
14
|
+
import { extractPriorTurnBaseline, pruneMultiTurnMessages } from './moa-context.js'
|
|
15
|
+
import { generateMoABenchmarkReport } from './moa-report.js'
|
|
15
16
|
|
|
16
17
|
// Re-exports for consumers & backward compatibility
|
|
17
18
|
export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
18
19
|
export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
19
20
|
export { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
*/
|
|
24
|
-
import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
25
|
-
export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel }
|
|
21
|
+
export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
22
|
+
export { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
|
23
|
+
export { streamMoATurn } from './moa-stream.js'
|
|
26
24
|
|
|
27
25
|
/**
|
|
28
26
|
* Executes the full Mixture of Agents pipeline.
|
|
@@ -39,18 +37,18 @@ export async function runMoAPipeline({
|
|
|
39
37
|
liveCanvas = null,
|
|
40
38
|
onStreamDelta = null,
|
|
41
39
|
signal = null,
|
|
40
|
+
testGateEnabled = null,
|
|
41
|
+
testCommand = null,
|
|
42
|
+
execFn = null,
|
|
42
43
|
}) {
|
|
43
44
|
if (signal?.aborted) return { error: new Error('Turn aborted') }
|
|
44
45
|
const startTime = Date.now()
|
|
45
46
|
|
|
46
47
|
// 1. Resolve configurations
|
|
47
48
|
const referenceModels = Array.isArray(preset?.reference_models) && preset.reference_models.length > 0
|
|
48
|
-
? preset.reference_models
|
|
49
|
-
: [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
|
|
50
|
-
|
|
49
|
+
? preset.reference_models : [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
|
|
51
50
|
const primaryJudge = preset?.aggregator?.provider && preset?.aggregator?.model
|
|
52
|
-
? preset.aggregator
|
|
53
|
-
: { provider: 'codex', model: 'gpt-5.6-sol' }
|
|
51
|
+
? preset.aggregator : { provider: 'codex', model: 'gpt-5.6-sol' }
|
|
54
52
|
|
|
55
53
|
const fallbackJudges = Array.isArray(preset?.aggregator_fallbacks) ? preset.aggregator_fallbacks : []
|
|
56
54
|
const judgesChain = [primaryJudge, ...fallbackJudges]
|
|
@@ -60,17 +58,44 @@ export async function runMoAPipeline({
|
|
|
60
58
|
const maxTokens = typeof preset?.max_tokens === 'number' ? preset.max_tokens : 4096
|
|
61
59
|
const judgeCriteria = preset?.judge_criteria || ''
|
|
62
60
|
const isFastMode = referenceModels.length === 1 && !preset?.curator_synthesis
|
|
63
|
-
const isCuratorSynthesis = Boolean(preset?.curator_synthesis)
|
|
64
|
-
const
|
|
65
|
-
const
|
|
66
|
-
const gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
|
|
67
|
-
const candidateRetries = 1
|
|
68
|
-
const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
61
|
+
const isCuratorSynthesis = Boolean(preset?.curator_synthesis), isStreamAggregator = preset?.stream_aggregator !== false
|
|
62
|
+
const isQuorumEnabled = Boolean(preset?.quorum_enabled), gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
|
|
63
|
+
const candidateRetries = 1, refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
69
64
|
const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
|
|
70
|
-
const isBlindEvaluation = Boolean(preset?.blind_evaluation)
|
|
71
|
-
const
|
|
72
|
-
|
|
73
|
-
|
|
65
|
+
const isBlindEvaluation = Boolean(preset?.blind_evaluation), isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
|
|
66
|
+
const allowCandidateOverride = Boolean(preset?.allow_candidate_override), runId = crypto.randomUUID()
|
|
67
|
+
|
|
68
|
+
// Feature 6: Budget Guardrail check
|
|
69
|
+
const budgetCheck = applyBudgetGuardrails({
|
|
70
|
+
references: referenceModels,
|
|
71
|
+
enabled: preset?.budget_guard_enabled,
|
|
72
|
+
maxBudgetUsd: preset?.max_budget_usd,
|
|
73
|
+
action: preset?.budget_action || 'trim',
|
|
74
|
+
prices,
|
|
75
|
+
promptLength: userPrompt?.length || 1000,
|
|
76
|
+
})
|
|
77
|
+
if (!budgetCheck.allowed) {
|
|
78
|
+
return {
|
|
79
|
+
kind: 'failure',
|
|
80
|
+
content: `🛑 Budget Guardrail Abort: ${budgetCheck.reason}`,
|
|
81
|
+
aggregator: slotLabel(primaryJudge),
|
|
82
|
+
references: [],
|
|
83
|
+
presetName: preset?.name || 'default',
|
|
84
|
+
isRefinement: false,
|
|
85
|
+
isFastMode,
|
|
86
|
+
winningIndex: 0,
|
|
87
|
+
winnerModel: 'none',
|
|
88
|
+
promotedFiles: [],
|
|
89
|
+
usage: { totalTokens: 0, totalCostUsd: 0, candidates: [], aggregator: { totalTokens: 0, costUsd: 0 } },
|
|
90
|
+
skippedFiles: 0,
|
|
91
|
+
skippedList: [],
|
|
92
|
+
durationMs: Date.now() - startTime,
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
const effectiveReferences = budgetCheck.references
|
|
96
|
+
if (budgetCheck.action === 'trim' && typeof onProgress === 'function') {
|
|
97
|
+
onProgress(`✂️ *${budgetCheck.reason}*\n`)
|
|
98
|
+
}
|
|
74
99
|
|
|
75
100
|
// 2. Collect project context for refinement tasks
|
|
76
101
|
const collectedCtx = await collectProjectContext(cwd, 16000)
|
|
@@ -91,23 +116,28 @@ export async function runMoAPipeline({
|
|
|
91
116
|
})
|
|
92
117
|
}
|
|
93
118
|
|
|
94
|
-
// 3. Build prompts &
|
|
119
|
+
// 3. Build prompts & Feature 8 multi-turn context
|
|
120
|
+
const priorTurnBaseline = preset?.multi_turn_enabled !== false ? extractPriorTurnBaseline(messages) : null
|
|
121
|
+
const prunedHistory = preset?.multi_turn_enabled !== false ? pruneMultiTurnMessages(messages) : messages
|
|
122
|
+
|
|
95
123
|
const candidateSystemPrompt = projectContext
|
|
96
124
|
? `${REFERENCE_SYSTEM_PROMPT}\n\n### Current Project Files & Context:\n${projectContext}`
|
|
97
125
|
: REFERENCE_SYSTEM_PROMPT
|
|
98
126
|
|
|
99
127
|
const askClarifyingQuestions = preset?.ask_clarifying_questions !== false
|
|
100
128
|
const needsQuestions = askClarifyingQuestions && isBroadPromptRequiringQuestions(userPrompt, messages)
|
|
129
|
+
const enrichedMessages = [...prunedHistory, { role: 'user', content: userPrompt }]
|
|
101
130
|
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
// 4. Parallel fan-out to candidate models with quorum & transient retry
|
|
131
|
+
// 4. Parallel fan-out to candidate models with temperature gradient & local fallback
|
|
105
132
|
const referenceOutputs = await runReferencesParallel(
|
|
106
|
-
|
|
133
|
+
effectiveReferences,
|
|
107
134
|
enrichedMessages,
|
|
108
135
|
{
|
|
109
136
|
systemPrompt: candidateSystemPrompt,
|
|
110
137
|
temperature: refTemp,
|
|
138
|
+
temperature_gradient_enabled: Boolean(preset?.temperature_gradient_enabled),
|
|
139
|
+
local_fallback_enabled: Boolean(preset?.local_fallback_enabled),
|
|
140
|
+
local_fallback_models: preset?.local_fallback_models || [],
|
|
111
141
|
maxTokens,
|
|
112
142
|
prices,
|
|
113
143
|
timeoutSec: refTimeoutSec,
|
|
@@ -159,10 +189,20 @@ export async function runMoAPipeline({
|
|
|
159
189
|
const files = extractFileBlocks(ref.text)
|
|
160
190
|
ref.files = files
|
|
161
191
|
if (files.length > 0) {
|
|
192
|
+
ref.syntaxWarning = (verifyFileSyntax(files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
|
|
162
193
|
await writeCandidateWorkspace(cwd, i + 1, files)
|
|
163
194
|
}
|
|
164
195
|
}
|
|
165
196
|
|
|
197
|
+
// 5b. Test Execution Gate
|
|
198
|
+
await executeTestGateForCandidates({
|
|
199
|
+
cwd,
|
|
200
|
+
referenceOutputs,
|
|
201
|
+
preset,
|
|
202
|
+
options: { testGateEnabled, testCommand, execFn },
|
|
203
|
+
onProgress,
|
|
204
|
+
})
|
|
205
|
+
|
|
166
206
|
// 6. Questionnaire synthesis branch
|
|
167
207
|
if (needsQuestions) {
|
|
168
208
|
if (typeof onProgress === 'function') {
|
|
@@ -188,11 +228,7 @@ export async function runMoAPipeline({
|
|
|
188
228
|
qUsage = estimateTokenCost(primaryJudge, qFallbackUsage, prices)
|
|
189
229
|
} catch (err) {
|
|
190
230
|
console.warn('[dsh-moa] Questionnaire synthesis failed, proceeding with fallback questions:', err)
|
|
191
|
-
questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n
|
|
192
|
-
'1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n' +
|
|
193
|
-
'2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n' +
|
|
194
|
-
'3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n' +
|
|
195
|
-
'*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*'
|
|
231
|
+
questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*`
|
|
196
232
|
}
|
|
197
233
|
|
|
198
234
|
await cleanMoaWorkspaces(cwd)
|
|
@@ -344,12 +380,41 @@ export async function runMoAPipeline({
|
|
|
344
380
|
}
|
|
345
381
|
})
|
|
346
382
|
await Promise.allSettled(r2Promises)
|
|
383
|
+
await executeTestGateForCandidates({
|
|
384
|
+
cwd,
|
|
385
|
+
referenceOutputs: successfulRefs,
|
|
386
|
+
preset,
|
|
387
|
+
options: { testGateEnabled, testCommand, execFn },
|
|
388
|
+
onProgress,
|
|
389
|
+
})
|
|
347
390
|
}
|
|
348
391
|
|
|
392
|
+
// 7c. Feature 2: Multi-Judge Panel & Consensus Voting
|
|
393
|
+
let multiJudgeResult = null
|
|
394
|
+
if (preset?.multi_judge_enabled && successfulRefs.length > 1) {
|
|
395
|
+
multiJudgeResult = await executeMultiJudgePanel({
|
|
396
|
+
judges: (preset.judge_models && preset.judge_models.length > 0) ? preset.judge_models : [primaryJudge],
|
|
397
|
+
candidates: successfulRefs,
|
|
398
|
+
userPrompt,
|
|
399
|
+
callLlm,
|
|
400
|
+
strategy: preset.judge_voting_strategy || 'majority',
|
|
401
|
+
signal,
|
|
402
|
+
onProgress,
|
|
403
|
+
})
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
// 7d. Feature 3: Composite Block Merge Directive
|
|
407
|
+
const compositeMergeDirective = (preset?.composite_merge_enabled && successfulRefs.length > 1)
|
|
408
|
+
? buildCompositeBlockDirectives(successfulRefs)
|
|
409
|
+
: null
|
|
410
|
+
|
|
349
411
|
// 8. Synthesis phase via primary judge or fallback chain
|
|
350
412
|
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
351
413
|
curatorSynthesis: isCuratorSynthesis,
|
|
352
414
|
blindEvaluation: isBlindEvaluation,
|
|
415
|
+
priorTurnBaseline,
|
|
416
|
+
compositeMergeDirective,
|
|
417
|
+
consensusReport: multiJudgeResult?.consensusReport,
|
|
353
418
|
})
|
|
354
419
|
let synthesizedText = ''
|
|
355
420
|
let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
|
|
@@ -411,7 +476,10 @@ export async function runMoAPipeline({
|
|
|
411
476
|
}
|
|
412
477
|
|
|
413
478
|
// 9. Evaluate winner & promote files
|
|
414
|
-
const winningIndex =
|
|
479
|
+
const winningIndex = (multiJudgeResult?.consensus && multiJudgeResult?.winningCandidateIndex)
|
|
480
|
+
? multiJudgeResult.winningCandidateIndex
|
|
481
|
+
: parseWinnerIndex(synthesizedText, 1, referenceOutputs.length)
|
|
482
|
+
|
|
415
483
|
const recommendedAssembler = isCuratorSynthesis
|
|
416
484
|
? parseRecommendedAssembler(synthesizedText, winningIndex, referenceOutputs.length)
|
|
417
485
|
: null
|
|
@@ -434,13 +502,32 @@ export async function runMoAPipeline({
|
|
|
434
502
|
}
|
|
435
503
|
|
|
436
504
|
const livePreview = await createPromotedPreview(liveCanvas, cwd, promotedFiles, referenceOutputs)
|
|
437
|
-
|
|
438
505
|
const durationMs = Date.now() - startTime
|
|
439
506
|
const usageSummary = summarizeMoAUsage(referenceOutputs, aggUsage)
|
|
440
507
|
const winningRef = referenceOutputs[winningIndex - 1]
|
|
441
508
|
const winnerModel = winningRef?.label || slotLabel(referenceModels[0])
|
|
442
509
|
const finalAggLabel = slotLabel(chosenJudge)
|
|
443
510
|
|
|
511
|
+
// Feature 9: Benchmark & Post-Mortem PR report
|
|
512
|
+
let benchmarkReport = null
|
|
513
|
+
if (preset?.report_generation_enabled) {
|
|
514
|
+
benchmarkReport = generateMoABenchmarkReport({
|
|
515
|
+
runId,
|
|
516
|
+
timestamp: new Date().toISOString(),
|
|
517
|
+
preset: preset?.name || 'default',
|
|
518
|
+
prompt: userPrompt,
|
|
519
|
+
durationMs,
|
|
520
|
+
usage: usageSummary,
|
|
521
|
+
costUsd: usageSummary.totalCostUsd,
|
|
522
|
+
candidates: referenceOutputs,
|
|
523
|
+
winningIndex,
|
|
524
|
+
winningLabel: winnerModel,
|
|
525
|
+
consensus: multiJudgeResult,
|
|
526
|
+
testGate: referenceOutputs.find((r) => r.testResult)?.testResult || null,
|
|
527
|
+
isComposite: Boolean(preset?.composite_merge_enabled),
|
|
528
|
+
})
|
|
529
|
+
}
|
|
530
|
+
|
|
444
531
|
try {
|
|
445
532
|
recordMoaRun({
|
|
446
533
|
id: runId,
|
|
@@ -452,6 +539,8 @@ export async function runMoAPipeline({
|
|
|
452
539
|
winnerIndex: winningIndex,
|
|
453
540
|
winnerModel,
|
|
454
541
|
promotedFiles,
|
|
542
|
+
benchmarkReport,
|
|
543
|
+
consensus: multiJudgeResult,
|
|
455
544
|
totalTokens: usageSummary.totalTokens,
|
|
456
545
|
totalCostUsd: usageSummary.totalCostUsd,
|
|
457
546
|
durationMs,
|
|
@@ -475,6 +564,8 @@ export async function runMoAPipeline({
|
|
|
475
564
|
promotedFiles,
|
|
476
565
|
runId,
|
|
477
566
|
allowCandidateOverride,
|
|
567
|
+
benchmarkReport,
|
|
568
|
+
consensus: multiJudgeResult,
|
|
478
569
|
...(livePreview ? { liveCanvas: livePreview } : {}),
|
|
479
570
|
usage: usageSummary,
|
|
480
571
|
skippedFiles: collectedCtx?.skippedFiles || 0,
|
|
@@ -482,115 +573,3 @@ export async function runMoAPipeline({
|
|
|
482
573
|
durationMs,
|
|
483
574
|
}
|
|
484
575
|
}
|
|
485
|
-
|
|
486
|
-
/**
|
|
487
|
-
* Streams a full MoA turn into chat with live aggregator tokens and progress feedback.
|
|
488
|
-
*/
|
|
489
|
-
export async function* streamMoATurn({ targetPreset, userPrompt, messages, callLlm, cwd, prices, historyFilePath, liveCanvas }, options = {}) {
|
|
490
|
-
const signal = options?.signal
|
|
491
|
-
if (signal?.aborted) return
|
|
492
|
-
|
|
493
|
-
yield { type: 'block-start', index: 0, blockType: 'text' }
|
|
494
|
-
yield { type: 'text-delta', index: 0, text: '🧠 *Mixture of Agents started...*\n\n' }
|
|
495
|
-
|
|
496
|
-
// Async push-queue for zero-latency live delta streaming
|
|
497
|
-
const queue = []
|
|
498
|
-
let notify = null
|
|
499
|
-
|
|
500
|
-
const pushUpdate = (text) => {
|
|
501
|
-
queue.push({ type: 'text-delta', index: 0, text })
|
|
502
|
-
if (notify) {
|
|
503
|
-
notify()
|
|
504
|
-
notify = null
|
|
505
|
-
}
|
|
506
|
-
}
|
|
507
|
-
|
|
508
|
-
let done = false
|
|
509
|
-
const pipelinePromise = runMoAPipeline({
|
|
510
|
-
userPrompt,
|
|
511
|
-
messages,
|
|
512
|
-
preset: targetPreset,
|
|
513
|
-
callLlm,
|
|
514
|
-
cwd,
|
|
515
|
-
onProgress: pushUpdate,
|
|
516
|
-
onStreamDelta: (delta) => {
|
|
517
|
-
pushUpdate(delta)
|
|
518
|
-
},
|
|
519
|
-
prices,
|
|
520
|
-
historyFilePath,
|
|
521
|
-
liveCanvas,
|
|
522
|
-
signal,
|
|
523
|
-
})
|
|
524
|
-
.catch((err) => ({ error: err }))
|
|
525
|
-
.finally(() => {
|
|
526
|
-
done = true
|
|
527
|
-
if (notify) {
|
|
528
|
-
notify()
|
|
529
|
-
notify = null
|
|
530
|
-
}
|
|
531
|
-
})
|
|
532
|
-
|
|
533
|
-
const startTime = Date.now()
|
|
534
|
-
let lastYieldTime = Date.now()
|
|
535
|
-
|
|
536
|
-
try {
|
|
537
|
-
while (!done || queue.length > 0) {
|
|
538
|
-
if (signal?.aborted) {
|
|
539
|
-
await cleanMoaWorkspaces(cwd)
|
|
540
|
-
return
|
|
541
|
-
}
|
|
542
|
-
|
|
543
|
-
while (queue.length > 0) {
|
|
544
|
-
const item = queue.shift()
|
|
545
|
-
yield item
|
|
546
|
-
lastYieldTime = Date.now()
|
|
547
|
-
}
|
|
548
|
-
|
|
549
|
-
if (done) break
|
|
550
|
-
|
|
551
|
-
// Wait for next push item or max 2.5s heartbeat
|
|
552
|
-
await Promise.race([
|
|
553
|
-
new Promise((resolve) => { notify = resolve }),
|
|
554
|
-
new Promise((resolve) => { const t = setTimeout(resolve, 2500); t.unref?.(); }),
|
|
555
|
-
])
|
|
556
|
-
|
|
557
|
-
const elapsedSec = Math.floor((Date.now() - startTime) / 1000)
|
|
558
|
-
if (!done && Date.now() - lastYieldTime >= 3000) {
|
|
559
|
-
yield { type: 'text-delta', index: 0, text: `⏳ *[${elapsedSec}s] Still processing...*\n` }
|
|
560
|
-
lastYieldTime = Date.now()
|
|
561
|
-
}
|
|
562
|
-
}
|
|
563
|
-
|
|
564
|
-
const result = await pipelinePromise
|
|
565
|
-
if (signal?.aborted) {
|
|
566
|
-
await cleanMoaWorkspaces(cwd)
|
|
567
|
-
return
|
|
568
|
-
}
|
|
569
|
-
|
|
570
|
-
if (result?.error) {
|
|
571
|
-
const errText = '\n\n⚠️ **Mixture of Agents error**: ' + (result.error?.message || String(result.error))
|
|
572
|
-
yield { type: 'text-delta', index: 0, text: errText }
|
|
573
|
-
yield { type: 'block-end', index: 0, block: { type: 'text', text: errText } }
|
|
574
|
-
yield { type: 'finish', reason: { kind: 'stop' } }
|
|
575
|
-
return
|
|
576
|
-
}
|
|
577
|
-
|
|
578
|
-
const formatted = formatMoAResponse({
|
|
579
|
-
moaResult: result,
|
|
580
|
-
presetName: targetPreset?.name || 'default',
|
|
581
|
-
})
|
|
582
|
-
|
|
583
|
-
yield { type: 'text-delta', index: 0, text: '\n---\n\n' + formatted }
|
|
584
|
-
yield { type: 'block-end', index: 0, block: { type: 'text', text: formatted } }
|
|
585
|
-
yield {
|
|
586
|
-
type: 'usage',
|
|
587
|
-
usage: {
|
|
588
|
-
inputTokens: result?.usage?.totalTokens || 0,
|
|
589
|
-
outputTokens: Math.round((formatted.length || 0) / 4),
|
|
590
|
-
},
|
|
591
|
-
}
|
|
592
|
-
yield { type: 'finish', reason: { kind: 'stop' } }
|
|
593
|
-
} finally {
|
|
594
|
-
// cleanup
|
|
595
|
-
}
|
|
596
|
-
}
|