@goodandready/dsh-moa 0.2.10 → 0.2.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/moa-runner.js CHANGED
@@ -5,7 +5,7 @@
5
5
  * - Parallel fan-out to reference models with transient retry & quorum straggler mitigation
6
6
  * - Prompt caching aligned message structures
7
7
  * - Curator synthesis & antipatterns evaluation
8
- * - Aggregator fallback chain for resilience
8
+ * - Aggregator fallback chain for resilience & malformed output recovery
9
9
  * - Live token streaming for aggregator
10
10
  * - File promotion & Live Canvas sandbox preview
11
11
  */
@@ -20,311 +20,48 @@ import {
20
20
  cleanMoaWorkspaces,
21
21
  } from './file-workspace.js'
22
22
  import { estimateTokenCost } from './pricing.js'
23
- import { recordMoaRun } from './history.js'
23
+ import { recordMoaRun, recordMoaRunAsync } from './history.js'
24
+ import {
25
+ slotLabel,
26
+ cleanAdvisoryMessages,
27
+ isBroadPromptRequiringQuestions,
28
+ buildQuestionSynthesisPrompt,
29
+ buildCuratorSynthesisPrompt,
30
+ buildSynthesisPrompt,
31
+ SYSTEM_ROLE_PROPOSER,
32
+ ANTIPATTERNS_RUBRIC,
33
+ } from './moa-prompts.js'
34
+ import {
35
+ parseWinnerIndex,
36
+ parseRecommendedAssembler,
37
+ parseMoACommand,
38
+ stripOrSummarizeCode,
39
+ formatMoAResponse,
40
+ } from './moa-parser.js'
41
+
42
+ // Re-export prompt and parser functions for external consumers / backwards compatibility
43
+ export {
44
+ slotLabel,
45
+ cleanAdvisoryMessages,
46
+ isBroadPromptRequiringQuestions,
47
+ buildQuestionSynthesisPrompt,
48
+ buildCuratorSynthesisPrompt,
49
+ buildSynthesisPrompt,
50
+ ANTIPATTERNS_RUBRIC,
51
+ } from './moa-prompts.js'
52
+
53
+ export {
54
+ parseWinnerIndex,
55
+ parseRecommendedAssembler,
56
+ parseMoACommand,
57
+ stripOrSummarizeCode,
58
+ formatMoAResponse,
59
+ } from './moa-parser.js'
24
60
 
25
61
  export { estimateTokenCost } from './pricing.js'
26
62
 
27
63
  export const DEFAULT_PROMPT = 'Please propose an optimal, well-structured, production-ready solution with full code and explanations.'
28
-
29
- export const REFERENCE_SYSTEM_PROMPT = `You are an expert AI software architect and senior engineer acting as a candidate proposer in a Mixture of Agents (MoA) ensemble.
30
- Your task is to provide the highest-quality, robust, complete, production-ready solution to the user request.
31
- Write clean, modern, fully functional code without placeholders or shortcuts.
32
- When generating files for a project, explicitly specify file paths using fenced code blocks with file annotations, e.g.:
33
- \`\`\`html file="index.html"
34
- \`\`\`
35
- \`\`\`javascript file="script.js"
36
- \`\`\`
37
- \`\`\`css file="style.css"
38
- \`\`\``
39
-
40
- export const ANTIPATTERNS_RUBRIC = `### 🚫 Strict Antipatterns Evaluation Checklist
41
- Penalize and strictly downgrade candidates exhibiting any of the following flaws:
42
- 1. 🚫 Lazy Code & Placeholders:
43
- - Phrases like "// ... rest of code unchanged", "/* TODO: implement */", incomplete functions or stubs returning null/mock without notice.
44
- 2. 🚫 Blind Mocking:
45
- - Hardcoded dummy arrays instead of real dynamic logic, user input handling, or real API integration.
46
- 3. 🚫 Silent Failures & Missing Error Handling:
47
- - Missing try/catch around async calls, fetch, JSON.parse; lack of user-facing fallback or retry states.
48
- 4. 🚫 AI Slop UI & Poor Ergonomics:
49
- - Generic purple/cyan gradients on pure black backgrounds, blurry drop-shadows without borders, lack of typographic hierarchy, low contrast (e.g. light gray text on white).
50
- 5. 🚫 Missing UI States:
51
- - Lack of loading state (spinner/skeleton), error display with retry action, or empty state when no data exists.
52
- 6. 🚫 Broken Layout & Mobile Incompatibility:
53
- - Fixed pixel widths (e.g. width: 800px) overflowing small viewports; unscrollable modal dialogs.
54
- 7. 🚫 Monolithic God Objects & Overengineering:
55
- - Dumping 1000+ lines into a single unmaintainable file, or building 10+ abstraction layers for a 2-function task.
56
- 8. 🚫 Context Amnesia & Regressions:
57
- - Dropping or breaking previously functioning project features while adding new code.`
58
-
59
- export function slotLabel(slot) {
60
- if (!slot) return 'unknown'
61
- if (typeof slot === 'string') return slot
62
- if (slot.provider && slot.model) return `${slot.provider}:${slot.model}`
63
- return slot.model || slot.provider || 'unknown'
64
- }
65
-
66
- /**
67
- * Extracts and sanitizes conversation history for candidate models.
68
- * Robust against complex DSH message content types (strings, part arrays, objects, tool calls).
69
- */
70
- export function cleanAdvisoryMessages(messages = [], maxCharBudget = 24000) {
71
- if (!Array.isArray(messages)) return []
72
-
73
- const trimmed = []
74
- for (const msg of messages) {
75
- if (!msg || typeof msg !== 'object') continue
76
- const role = msg.role
77
- if (role !== 'user' && role !== 'assistant') continue
78
-
79
- let text = ''
80
- if (typeof msg.content === 'string') {
81
- text = msg.content
82
- } else if (Array.isArray(msg.content)) {
83
- text = msg.content
84
- .filter((part) => part && typeof part === 'object' && part.type === 'text' && typeof part.text === 'string')
85
- .map((part) => part.text)
86
- .join('\n')
87
- } else if (msg.content && typeof msg.content === 'object') {
88
- if (typeof msg.content.text === 'string') {
89
- text = msg.content.text
90
- } else if (typeof msg.content.content === 'string') {
91
- text = msg.content.content
92
- }
93
- } else if (typeof msg.text === 'string') {
94
- text = msg.text
95
- }
96
-
97
- text = text.trim()
98
- if (!text) continue
99
-
100
- if (text.length > maxCharBudget) {
101
- const head = text.slice(0, Math.floor(maxCharBudget * 0.7))
102
- const tail = text.slice(-Math.floor(maxCharBudget * 0.3))
103
- text = `${head}\n... [trimmed ${text.length - maxCharBudget} characters] ...\n${tail}`
104
- }
105
-
106
- trimmed.push({ role, content: text })
107
- }
108
- return trimmed
109
- }
110
-
111
- export function isBroadPromptRequiringQuestions(userPrompt = '', messages = []) {
112
- if (!userPrompt || typeof userPrompt !== 'string') return false
113
- const p = userPrompt.trim()
114
- const wordCount = p.split(/\s+/).length
115
-
116
- if (/^(да|нет|1|2|3|4|ок|погнали|давай|yes|no)\b/i.test(p) && wordCount <= 5) {
117
- return false
118
- }
119
-
120
- const hasRecentQuestion = messages.some((m) => {
121
- const text = typeof m?.content === 'string' ? m.content : JSON.stringify(m?.content || '')
122
- return text.includes('Уточнение требований') || text.includes('опросник') || text.includes('Вариант 1')
123
- })
124
- if (hasRecentQuestion) {
125
- return false
126
- }
127
-
128
- const creationTriggers = [
129
- 'сделай', 'создай', 'напиши', 'разработай', 'придумай', 'реализуй',
130
- 'make', 'build', 'create', 'generate', 'develop',
131
- ]
132
- const startsWithCreation = creationTriggers.some((t) => p.toLowerCase().startsWith(t))
133
-
134
- if (startsWithCreation && wordCount <= 18) {
135
- return true
136
- }
137
-
138
- const vagueNouns = ['приложение', 'игру', 'сервис', 'сайт', 'лендинг', 'калькулятор', 'виджет', 'дашборд', 'app', 'game', 'tool', 'website']
139
- if (vagueNouns.some((n) => p.toLowerCase().includes(n))) {
140
- if (wordCount <= 12) return true
141
- }
142
-
143
- return false
144
- }
145
-
146
- export function buildQuestionSynthesisPrompt(userPrompt, referenceOutputs = []) {
147
- const joined = referenceOutputs
148
- .map((r, i) => `Advisor ${i + 1} (${r.label}):\n${r.text}`)
149
- .join('\n\n')
150
-
151
- return `You are the lead architect and judge in a Mixture of Agents (MoA) ensemble.
152
- The user gave the task:
153
- "${userPrompt}"
154
-
155
- The advisors proposed the following decision points and clarifications:
156
- ${joined}
157
-
158
- Your task is to synthesize a single, compact, friendly and structured questionnaire (2-4 questions) in the same language as the user's prompt.
159
- Each question must offer 2-3 concrete recommended answer options (e.g.: 1. Format: single-file HTML/JS or React? 2. Style: minimalism, iOS or neubrutalism?).
160
- At the end, add a note that the user can answer briefly (e.g.: "1, 2, dark theme") or trust the defaults.`
161
- }
162
-
163
- export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '') {
164
- const joined = referenceOutputs
165
- .map((r, i) => {
166
- const fileSummary = (r.files && r.files.length > 0)
167
- ? ` [Files created: ${r.files.map((f) => f.relativePath).join(', ')}]`
168
- : ''
169
- let textContent = r.text
170
- if (referenceOutputs.length >= 3 && textContent.length > 3000) {
171
- textContent = stripOrSummarizeCode(textContent)
172
- }
173
- return `Candidate ${i + 1} — ${r.label}:${fileSummary}\n${textContent}`
174
- })
175
- .join('\n\n')
176
-
177
- const criteriaBlock = judgeCriteria && judgeCriteria.trim()
178
- ? `\n### 🎯 Additional evaluation criteria:\n${judgeCriteria.trim()}\n`
179
- : ''
180
-
181
- return `You are the expert Lead Technical Curator and Solution Architect in a Mixture of Agents (MoA) ensemble.
182
- Your mission is not merely to select one candidate, but to synthesize the optimal solution by extracting the finest components from each candidate's response, identifying potential flaws using the strict antipatterns rubric, and selecting/advising which single agent model is best suited to assemble the final unified deliverable.
183
-
184
- User Request:
185
- ${userPrompt}
186
- ${criteriaBlock}
187
- Candidate proposals:
188
- ${joined}
189
-
190
- ${ANTIPATTERNS_RUBRIC}
191
-
192
- Instructions:
193
- Your response MUST be structured into three clear parts (respond in the same language as the user's prompt, e.g. Russian):
194
-
195
- ### 1. 🔍 Curator Analysis & Component Breakdown
196
- - For EACH candidate, provide:
197
- - ⭐ **Strongest aspects** (e.g. robust architecture, superior UI/CSS design, clean data validation).
198
- - ⚠️ **Defects or Antipatterns** found from the checklist above.
199
- - Highlight which candidate provides the best foundation for each component (e.g. Candidate 1 for core logic, Candidate 2 for visual UI).
200
-
201
- ### 2. 🧩 Assembly Recipe & Recommended Master Assembler
202
- - Recommend the best single agent model to assemble and finalize the solution:
203
- RECOMMENDED_ASSEMBLER: <number from 1 to N> (<provider:model>)
204
- - State the machine winner index marker for file promotion:
205
- WINNER_CANDIDATE_INDEX: <number from 1 to N>
206
- - Provide the exact blueprint / instructions for combining the best pieces into a unified deliverable.
207
-
208
- ### 3. 🚀 Unified Solution & Execution Guide
209
- - Present the final synthesized code or complete instructions combining the best candidate features.
210
- - How to run, verify, and use the deliverable.`
211
- }
212
-
213
- export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
214
- if (options.curatorSynthesis) {
215
- return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria)
216
- }
217
-
218
- const joined = referenceOutputs
219
- .map((r, i) => {
220
- const fileSummary = (r.files && r.files.length > 0)
221
- ? ` [Files created: ${r.files.map((f) => f.relativePath).join(', ')}]`
222
- : ''
223
- let textContent = r.text
224
- if (referenceOutputs.length >= 3 && textContent.length > 3000) {
225
- textContent = stripOrSummarizeCode(textContent)
226
- }
227
- return `Reference ${i + 1} — ${r.label}:${fileSummary}\n${textContent}`
228
- })
229
- .join('\n\n')
230
-
231
- const criteriaBlock = judgeCriteria && judgeCriteria.trim()
232
- ? `\n### 🎯 Additional evaluation criteria from the user:\n${judgeCriteria.trim()}\n`
233
- : ''
234
-
235
- return `You are the expert aggregator/judge in a Mixture of Agents (MoA) process. You evaluate solutions from multiple candidate models, judge which one is best (or how to combine their best parts), and deliver the final authoritative verdict and solution.
236
-
237
- Original user prompt:
238
- ${userPrompt}
239
- ${criteriaBlock}
240
- Reference responses from candidate models:
241
- ${joined}
242
-
243
- ${ANTIPATTERNS_RUBRIC}
244
-
245
- Instructions:
246
- Your response MUST be structured into three clear parts (respond in the same language as the user's prompt, e.g. Russian):
247
-
248
- ### 1. ⚖️ Judge verdict and comparative analysis
249
- - **Winner**: clearly name the model and candidate number (e.g. "Winner: Candidate 1 (opencode-go:deepseek-v4-flash)" or "Reference 1 (label) is chosen").
250
- - Always add the machine winner-selection marker:
251
- WINNER_CANDIDATE_INDEX: <number from 1 to N>
252
- - **Why this choice**: compare code, architecture, strengths, weaknesses and reliability of all candidates in detail.
253
-
254
- ### 2. 📁 Project files created
255
- - List the winner files promoted to the project root and their purpose.
256
-
257
- ### 3. 🚀 How to run and use
258
- - Describe how to open and run the created project.`
259
- }
260
-
261
- export function parseWinnerIndex(judgeText, defaultIndex = 1, candidateCount = Infinity) {
262
- if (!judgeText || typeof judgeText !== 'string') return defaultIndex
263
- const inRange = (idx) => !isNaN(idx) && idx >= 1 && idx <= candidateCount
264
- const match = /WINNER_CANDIDATE_INDEX:\s*(\d+)/i.exec(judgeText)
265
- if (match) {
266
- const idx = parseInt(match[1], 10)
267
- if (inRange(idx)) return idx
268
- }
269
- const candMatch = /(?:Кандидат|Candidate|Reference)\s*(\d+)\b/i.exec(judgeText)
270
- if (candMatch) {
271
- const idx = parseInt(candMatch[1], 10)
272
- if (inRange(idx)) return idx
273
- }
274
- return defaultIndex
275
- }
276
-
277
- export function parseRecommendedAssembler(judgeText, defaultIndex = 1, candidateCount = Infinity) {
278
- if (!judgeText || typeof judgeText !== 'string') return { index: defaultIndex, label: '' }
279
- const inRange = (idx) => !isNaN(idx) && idx >= 1 && idx <= candidateCount
280
- const match = /RECOMMENDED_ASSEMBLER:\s*(\d+)(?:\s*\(([^)]+)\))?/i.exec(judgeText)
281
- if (match) {
282
- const idx = parseInt(match[1], 10)
283
- if (inRange(idx)) {
284
- return { index: idx, label: match[2]?.trim() || '' }
285
- }
286
- }
287
- return { index: defaultIndex, label: '' }
288
- }
289
-
290
- export function parseMoACommand(text, presets = []) {
291
- if (typeof text !== 'string' || !text.startsWith('/moa')) {
292
- return null
293
- }
294
-
295
- const remainder = text.slice(4).trim()
296
- if (!remainder) {
297
- return {
298
- presetName: 'default',
299
- prompt: '',
300
- }
301
- }
302
-
303
- const matchPreset = /^--preset(?:=|\s+)([a-zA-Z0-9_-]+)\s*(.*)/s.exec(remainder)
304
- if (matchPreset) {
305
- return {
306
- presetName: matchPreset[1],
307
- prompt: matchPreset[2] || '',
308
- }
309
- }
310
-
311
- const parts = remainder.split(/\s+/)
312
- const firstWord = parts[0]
313
- if (Array.isArray(presets)) {
314
- const matched = presets.find((p) => p.name === firstWord)
315
- if (matched) {
316
- return {
317
- presetName: matched.name,
318
- prompt: parts.slice(1).join(' ').trim(),
319
- }
320
- }
321
- }
322
-
323
- return {
324
- presetName: 'default',
325
- prompt: remainder,
326
- }
327
- }
64
+ export const REFERENCE_SYSTEM_PROMPT = SYSTEM_ROLE_PROPOSER
328
65
 
329
66
  /**
330
67
  * Invokes LLM call with transient retry for recoverable network/rate-limit errors.
@@ -357,9 +94,9 @@ export async function runReferencesParallel(references, messages, options = {},
357
94
 
358
95
  const advisoryMessages = cleanAdvisoryMessages(messages)
359
96
  const systemPrompt = options.systemPrompt || REFERENCE_SYSTEM_PROMPT
360
- // Deterministic prefix for optimal prompt caching hit rate (#2.3)
97
+ // Deterministic prefix for optimal prompt caching hit rate
361
98
  const fullMessages = [{ role: 'system', content: systemPrompt }, ...advisoryMessages]
362
- const timeoutMs = options.timeoutMs ?? 75000
99
+ const timeoutMs = options.timeoutMs ?? ((options.timeoutSec ?? 60) * 1000)
363
100
  const maxRetries = options.maxRetries ?? 0
364
101
  const quorumEnabled = Boolean(options.quorumEnabled)
365
102
  const gracePeriodMs = (options.gracePeriodSec ?? 10) * 1000
@@ -391,14 +128,16 @@ export async function runReferencesParallel(references, messages, options = {},
391
128
  messages: fullMessages,
392
129
  temperature: options.temperature ?? 0.6,
393
130
  maxTokens: options.maxTokens ?? 4096,
131
+ timeoutMs,
394
132
  signal: abortControllers[i].signal,
395
133
  },
396
134
  maxRetries
397
135
  )
398
136
 
399
137
  const timeoutPromise = new Promise((_, reject) => {
400
- const timer = setTimeout(() => reject(new Error(`Timeout after ${timeoutMs / 1000}s`)), timeoutMs)
138
+ const timer = setTimeout(() => reject(new Error(`Timeout after ${Math.round(timeoutMs / 1000)}s`)), timeoutMs)
401
139
  timer.unref?.()
140
+ abortControllers[i].signal.addEventListener('abort', () => clearTimeout(timer), { once: true })
402
141
  })
403
142
 
404
143
  const res = await Promise.race([callPromise, timeoutPromise])
@@ -412,7 +151,7 @@ export async function runReferencesParallel(references, messages, options = {},
412
151
 
413
152
  finishedCount++
414
153
  if (typeof onProgress === 'function') {
415
- const costStr = costInfo.costUsd > 0 ? ` (~\${costInfo.costUsd.toFixed(4)})` : ''
154
+ const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
416
155
  onProgress(`✅ *Candidate ${i + 1}/${total} (${label}) finished${costStr}*\n`)
417
156
  }
418
157
 
@@ -450,22 +189,22 @@ export async function runReferencesParallel(references, messages, options = {},
450
189
  runOne()
451
190
  })
452
191
 
453
- // Quorum straggler mitigation: exit as soon as grace period expires
192
+ // Straggler mitigation via Quorum + Grace Period
454
193
  if (quorumEnabled && total >= 3) {
455
194
  const quorumTarget = Math.max(2, Math.min(total - 1, Math.ceil(total * 0.6)))
456
195
  let graceTimer = null
457
- let quorumTriggered = false
458
196
 
459
197
  await new Promise((resolve) => {
460
198
  const checkQuorum = () => {
461
199
  if (finishedCount >= total) {
462
200
  if (graceTimer) clearTimeout(graceTimer)
463
- return resolve()
201
+ resolve()
202
+ return
464
203
  }
465
- if (finishedCount >= quorumTarget && !quorumTriggered) {
466
- quorumTriggered = true
204
+
205
+ if (finishedCount >= quorumTarget && !graceTimer) {
467
206
  if (typeof onProgress === 'function') {
468
- onProgress(`⏳ *Quorum reached (${finishedCount}/${total}). Grace period ${gracePeriodMs / 1000}s for remaining models...*\n`)
207
+ onProgress(`⏳ *Quorum reached (${finishedCount}/${total}). Starting ${Math.round(gracePeriodMs / 1000)}s grace period for stragglers...*\n`)
469
208
  }
470
209
  graceTimer = setTimeout(() => {
471
210
  if (typeof onProgress === 'function' && finishedCount < total) {
@@ -478,7 +217,7 @@ export async function runReferencesParallel(references, messages, options = {},
478
217
  }
479
218
  resolve()
480
219
  }, gracePeriodMs)
481
- // active timer keeps event loop alive
220
+ graceTimer.unref?.()
482
221
  }
483
222
  }
484
223
 
@@ -548,7 +287,7 @@ async function createPromotedPreview(liveCanvas, cwd, promotedFiles, referenceOu
548
287
  }
549
288
 
550
289
  /**
551
- * Main MoA pipeline runner with Curator Synthesis, Fallback Chains, and Live Token Streaming.
290
+ * Executes the full Mixture of Agents pipeline.
552
291
  */
553
292
  export async function runMoAPipeline({
554
293
  userPrompt,
@@ -556,57 +295,59 @@ export async function runMoAPipeline({
556
295
  preset,
557
296
  callLlm,
558
297
  cwd = process.cwd(),
559
- onProgress,
560
- onStreamDelta,
561
298
  prices = {},
562
- historyFilePath,
299
+ onProgress = null,
300
+ historyFilePath = null,
563
301
  liveCanvas = null,
302
+ onStreamDelta = null,
564
303
  }) {
565
304
  const startTime = Date.now()
566
- const referenceModels = preset?.reference_models || [
567
- { provider: 'opencode-go', model: 'deepseek-v4-flash' },
568
- { provider: 'grok', model: 'grok-build-0.1' },
569
- ]
570
- const primaryJudge = preset?.aggregator || { provider: 'codex', model: 'gpt-5.6-sol' }
305
+
306
+ // 1. Resolve configurations
307
+ const referenceModels = Array.isArray(preset?.reference_models) && preset.reference_models.length > 0
308
+ ? preset.reference_models
309
+ : [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
310
+
311
+ const primaryJudge = preset?.aggregator?.provider && preset?.aggregator?.model
312
+ ? preset.aggregator
313
+ : { provider: 'codex', model: 'gpt-5.6-sol' }
314
+
571
315
  const fallbackJudges = Array.isArray(preset?.aggregator_fallbacks) ? preset.aggregator_fallbacks : []
572
316
  const judgesChain = [primaryJudge, ...fallbackJudges]
573
317
 
574
- const refTemp = preset?.reference_temperature ?? 0.6
575
- const aggTemp = preset?.aggregator_temperature ?? 0.4
576
- const maxTokens = preset?.max_tokens ?? 4096
577
- const judgeCriteria = preset?.judge_criteria ?? ''
578
- const isFastMode = referenceModels.length === 1
318
+ const refTemp = typeof preset?.reference_temperature === 'number' ? preset.reference_temperature : 0.6
319
+ const aggTemp = typeof preset?.aggregator_temperature === 'number' ? preset.aggregator_temperature : 0.4
320
+ const maxTokens = typeof preset?.max_tokens === 'number' ? preset.max_tokens : 4096
321
+ const judgeCriteria = preset?.judge_criteria || ''
322
+ const isFastMode = referenceModels.length === 1 && !preset?.curator_synthesis
579
323
  const isCuratorSynthesis = Boolean(preset?.curator_synthesis)
580
324
  const isStreamAggregator = preset?.stream_aggregator !== false
581
325
  const isQuorumEnabled = Boolean(preset?.quorum_enabled)
582
- const gracePeriodSec = preset?.grace_period_sec ?? 10
583
- const candidateRetries = preset?.retry_count ?? 1
584
-
585
- // 1. Collect current project workspace context
586
- if (typeof onProgress === 'function') {
587
- onProgress('🔍 *Scanning project context...*\n')
588
- }
589
- const projectContext = await collectProjectContext(cwd)
590
- const isRefinement = isRefinementTask(userPrompt, projectContext.files)
591
-
592
- // 2. Check if broad prompt requires user questionnaire
593
- const askQuestions = preset?.ask_clarifying_questions !== false
594
- const needsQuestions = !isFastMode && askQuestions && !isRefinement && isBroadPromptRequiringQuestions(userPrompt, messages)
595
-
596
- // 3. Build enriched prompt for candidates
597
- let candidateSystemPrompt = REFERENCE_SYSTEM_PROMPT
598
- if (isRefinement && projectContext.files.length > 0) {
599
- const fileList = projectContext.files.map((f) => `- \`${f.relativePath}\` (${f.content.length} chars)`).join('\n')
600
- const fileContents = projectContext.files.map((f) => `### File: ${f.relativePath}\n\`\`\`\n${f.content}\n\`\`\``).join('\n\n')
601
- candidateSystemPrompt += `\n\nExisting project structure:\n${fileList}\n\nProject files:\n${fileContents}\n\nYou are modifying an existing project. Output modified or new files with explicit file paths.`
326
+ const gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
327
+ const candidateRetries = 1
328
+ const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
329
+ const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
330
+ const isBlindEvaluation = Boolean(preset?.blind_evaluation)
331
+
332
+ // 2. Collect project context for refinement tasks
333
+ const isRefinement = isRefinementTask(userPrompt)
334
+ let projectContext = ''
335
+ if (isRefinement) {
336
+ if (typeof onProgress === 'function') {
337
+ onProgress('🔍 *Reading project files for refinement context...*\n')
338
+ }
339
+ projectContext = await collectProjectContext(cwd, 16000)
602
340
  }
603
341
 
604
- let promptForCandidates = userPrompt
605
- if (needsQuestions) {
606
- promptForCandidates += "\n(Note: the task is broad. Propose the key architectural and functional decision points to clarify the user's requirements.)"
607
- }
342
+ // 3. Build prompts & evaluate broad questionnaire needs
343
+ const candidateSystemPrompt = projectContext
344
+ ? `${REFERENCE_SYSTEM_PROMPT}\n\n### Current Project Files & Context:\n${projectContext}`
345
+ : REFERENCE_SYSTEM_PROMPT
346
+
347
+ const askClarifyingQuestions = preset?.ask_clarifying_questions !== false
348
+ const needsQuestions = askClarifyingQuestions && isBroadPromptRequiringQuestions(userPrompt, messages)
608
349
 
609
- const enrichedMessages = [...messages, { role: 'user', content: promptForCandidates }]
350
+ const enrichedMessages = [...messages, { role: 'user', content: userPrompt }]
610
351
 
611
352
  // 4. Parallel fan-out to candidate models with quorum & transient retry
612
353
  const referenceOutputs = await runReferencesParallel(
@@ -617,6 +358,7 @@ export async function runMoAPipeline({
617
358
  temperature: refTemp,
618
359
  maxTokens,
619
360
  prices,
361
+ timeoutSec: refTimeoutSec,
620
362
  quorumEnabled: isQuorumEnabled,
621
363
  gracePeriodSec,
622
364
  maxRetries: candidateRetries,
@@ -676,21 +418,27 @@ export async function runMoAPipeline({
676
418
  messages: [{ role: 'user', content: questionPrompt }],
677
419
  temperature: 0.3,
678
420
  maxTokens: 2048,
421
+ timeoutMs: aggTimeoutSec * 1000,
679
422
  }, 1)
680
423
  questionsContent = typeof qRes === 'string' ? qRes : (qRes?.content || qRes?.text || '')
681
- const qFallback = (typeof qRes === 'object' && qRes?.usage) ? qRes.usage : {
424
+ const qFallbackUsage = (typeof qRes === 'object' && qRes?.usage) ? qRes.usage : {
682
425
  inputTokens: Math.max(1, Math.round(questionPrompt.length / 4)),
683
426
  outputTokens: Math.max(1, Math.round(questionsContent.length / 4)),
684
427
  }
685
- qUsage = estimateTokenCost(primaryJudge, qFallback, prices)
686
- } catch {
687
- questionsContent = '### Project requirements clarification\nPlease specify the implementation details and the desired stack.'
428
+ qUsage = estimateTokenCost(primaryJudge, qFallbackUsage, prices)
429
+ } catch (err) {
430
+ console.warn('[dsh-moa] Questionnaire synthesis failed, proceeding with fallback questions:', err)
431
+ questionsContent = `### Уточнение требований по задаче: "${userPrompt}"\n\n` +
432
+ '1. **Формат решения**: однофайловый HTML/JS или многомодульный проект?\n' +
433
+ '2. **Стиль и визуальное оформление**: минимализм, темная тема или нейтральный интерфейс?\n' +
434
+ '3. **Функциональные приоритеты**: базовый MVP или расширенная реализация?\n\n' +
435
+ '*Ответьте кратко (например, "1, 2") или доверьтесь выбору по умолчанию.*'
688
436
  }
689
437
 
690
438
  await cleanMoaWorkspaces(cwd)
691
-
692
- const qTokens = referenceOutputs.reduce((acc, r) => acc + (r.usage?.totalTokens || 0), 0) + qUsage.totalTokens
693
- const qCostUsd = Number((referenceOutputs.reduce((acc, r) => acc + (r.costUsd || 0), 0) + qUsage.costUsd).toFixed(5))
439
+ const totalTokens = referenceOutputs.reduce((acc, r) => acc + (r.usage?.totalTokens || 0), 0) + qUsage.totalTokens
440
+ const totalCostUsd = Number((referenceOutputs.reduce((acc, r) => acc + (r.costUsd || 0), 0) + qUsage.costUsd).toFixed(5))
441
+ const durationMs = Date.now() - startTime
694
442
 
695
443
  try {
696
444
  recordMoaRun({
@@ -698,41 +446,51 @@ export async function runMoAPipeline({
698
446
  preset: preset?.name || 'default',
699
447
  isRefinement,
700
448
  candidates: candidatesForHistory(referenceOutputs),
701
- aggregator: null,
449
+ aggregator: { provider: primaryJudge.provider, model: primaryJudge.model, usage: qUsage, costUsd: qUsage.costUsd },
702
450
  winnerIndex: -1,
703
- winnerModel: '',
451
+ winnerModel: 'none',
704
452
  promotedFiles: [],
705
- totalTokens: qTokens,
706
- totalCostUsd: qCostUsd,
707
- durationMs: Date.now() - startTime,
708
- }, historyFilePath)
709
- } catch (histErr) {
710
- console.warn('[dsh-moa] Failed to record questionnaire run in history:', histErr)
711
- }
453
+ totalTokens,
454
+ totalCostUsd,
455
+ durationMs,
456
+ }, historyFilePath || undefined)
457
+ } catch {}
712
458
 
713
459
  return {
714
460
  kind: 'questions',
715
461
  content: questionsContent,
462
+ aggregator: slotLabel(primaryJudge),
716
463
  references: referenceOutputs,
717
464
  presetName: preset?.name || 'default',
718
- isRefinement: false,
465
+ isRefinement,
466
+ isFastMode,
467
+ winningIndex: 0,
468
+ winnerModel: 'none',
469
+ promotedFiles: [],
470
+ usage: {
471
+ totalTokens,
472
+ totalCostUsd,
473
+ candidates: referenceOutputs.map((r) => ({ label: r.label, usage: r.usage, costUsd: r.costUsd })),
474
+ aggregator: qUsage,
475
+ },
476
+ durationMs,
719
477
  }
720
478
  }
721
479
 
722
- // 7. Fast mode bypass for single candidate
480
+ // 7. Fast Mode (1 candidate, bypass judge)
723
481
  if (isFastMode) {
724
- const single = referenceOutputs[0]
482
+ const single = successfulRefs[0]
725
483
  let promotedFiles = []
726
- if (single.files && single.files.length > 0) {
727
- promotedFiles = await promoteCandidateWorkspace(cwd, 1)
484
+ if (single.files?.length > 0) {
485
+ promotedFiles = await promoteCandidateWorkspace(cwd, single.index)
728
486
  } else {
729
487
  await cleanMoaWorkspaces(cwd)
730
488
  }
731
489
 
490
+ const livePreview = await createPromotedPreview(liveCanvas, cwd, promotedFiles, referenceOutputs)
732
491
  const totalTokens = single.usage?.totalTokens || 0
733
- const totalCostUsd = single.costUsd || 0
492
+ const totalCostUsd = Number((single.costUsd || 0).toFixed(5))
734
493
  const durationMs = Date.now() - startTime
735
- const livePreview = await createPromotedPreview(liveCanvas, cwd, promotedFiles, referenceOutputs)
736
494
 
737
495
  try {
738
496
  recordMoaRun({
@@ -742,16 +500,14 @@ export async function runMoAPipeline({
742
500
  isFastMode: true,
743
501
  candidates: candidatesForHistory(referenceOutputs),
744
502
  aggregator: null,
745
- winnerIndex: 1,
503
+ winnerIndex: single.index,
746
504
  winnerModel: single.label,
747
505
  promotedFiles,
748
506
  totalTokens,
749
507
  totalCostUsd,
750
508
  durationMs,
751
- }, historyFilePath)
752
- } catch (histErr) {
753
- console.warn('[dsh-moa] Failed to record fast-mode run in history:', histErr)
754
- }
509
+ }, historyFilePath || undefined)
510
+ } catch {}
755
511
 
756
512
  return {
757
513
  kind: 'synthesis',
@@ -761,28 +517,25 @@ export async function runMoAPipeline({
761
517
  presetName: preset?.name || 'default',
762
518
  isRefinement,
763
519
  isFastMode: true,
764
- winningIndex: 1,
520
+ winningIndex: single.index,
765
521
  winnerModel: single.label,
766
522
  promotedFiles,
767
523
  ...(livePreview ? { liveCanvas: livePreview } : {}),
768
524
  usage: {
769
525
  totalTokens,
770
526
  totalCostUsd,
771
- candidates: [{ label: single.label, usage: single.usage, costUsd: single.costUsd }],
527
+ candidates: referenceOutputs.map((r) => ({ label: r.label, usage: r.usage, costUsd: r.costUsd })),
772
528
  aggregator: { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 },
773
529
  },
774
530
  durationMs,
775
531
  }
776
532
  }
777
533
 
778
- // 8. Aggregator / Judge synthesis with fallback chain (#3.3) & streaming (#2.1)
779
- const synthesisPrompt = buildSynthesisPrompt(
780
- userPrompt,
781
- referenceOutputs,
782
- judgeCriteria,
783
- { curatorSynthesis: isCuratorSynthesis }
784
- )
785
-
534
+ // 8. Synthesis phase via primary judge or fallback chain
535
+ const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
536
+ curatorSynthesis: isCuratorSynthesis,
537
+ blindEvaluation: isBlindEvaluation,
538
+ })
786
539
  let synthesizedText = ''
787
540
  let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
788
541
  let chosenJudge = primaryJudge
@@ -794,9 +547,9 @@ export async function runMoAPipeline({
794
547
  const currentLabel = slotLabel(currentJudge)
795
548
 
796
549
  if (typeof onProgress === 'function') {
797
- const judgeTitle = isCuratorSynthesis ? 'Lead Curator' : 'Judge'
550
+ const modeTitle = isCuratorSynthesis ? 'Curator' : 'Judge'
798
551
  const fallbackBadge = jIdx > 0 ? ` (Fallback #${jIdx})` : ''
799
- onProgress(`⚖️ *${judgeTitle} (${currentLabel})${fallbackBadge} evaluates candidates and synthesizes the solution...*\n`)
552
+ onProgress(`⚖️ *${modeTitle} (${currentLabel}${fallbackBadge}) synthesizing solution...*\n`)
800
553
  }
801
554
 
802
555
  try {
@@ -806,6 +559,7 @@ export async function runMoAPipeline({
806
559
  messages: [{ role: 'user', content: synthesisPrompt }],
807
560
  temperature: aggTemp,
808
561
  maxTokens,
562
+ timeoutMs: aggTimeoutSec * 1000,
809
563
  onStreamDelta: (delta) => {
810
564
  if (isStreamAggregator && typeof onStreamDelta === 'function') {
811
565
  onStreamDelta(delta)
@@ -814,6 +568,7 @@ export async function runMoAPipeline({
814
568
  }, 0)
815
569
 
816
570
  synthesizedText = typeof aggRes === 'string' ? aggRes : (aggRes?.content || aggRes?.text || '')
571
+
817
572
  const aggFallbackUsage = (typeof aggRes === 'object' && aggRes?.usage) ? aggRes.usage : {
818
573
  inputTokens: Math.max(1, Math.round(synthesisPrompt.length / 4)),
819
574
  outputTokens: Math.max(1, Math.round(synthesizedText.length / 4)),
@@ -885,7 +640,7 @@ export async function runMoAPipeline({
885
640
  totalTokens,
886
641
  totalCostUsd,
887
642
  durationMs,
888
- }, historyFilePath)
643
+ }, historyFilePath || undefined)
889
644
  } catch (histErr) {
890
645
  console.warn('[dsh-moa] Failed to record run in history:', histErr)
891
646
  }
@@ -915,103 +670,8 @@ export async function runMoAPipeline({
915
670
  }
916
671
 
917
672
  /**
918
- * Replaces large code blocks with concise file/code summaries to avoid token waste and huge chat dumps.
673
+ * Streams a full MoA turn into chat with live aggregator tokens and progress feedback.
919
674
  */
920
- export function stripOrSummarizeCode(text) {
921
- if (!text || typeof text !== 'string') return ''
922
- return text.replace(/```([a-zA-Z0-9_\-\.\/]*)\s*([\w\.\/\-]+\.[a-zA-Z0-9]+)?\n([\s\S]*?)```/g, (match, lang, fileTag, code) => {
923
- const lines = code.trim().split('\n')
924
- if (lines.length <= 3 && !/html|jsx|tsx|vue|svelte|css|js|ts/i.test(lang)) {
925
- return match
926
- }
927
- const fileHint = fileTag || (code.match(/^\s*(?:\/\/|#|<!--|\/\*)\s*(?:file|filepath|path):\s*([^\s*]+)/im)?.[1])
928
- const label = fileHint ? `file \`${fileHint}\`` : (lang ? `code \`${lang}\`` : 'code')
929
- return `\n> 📄 *[${label} - ${lines.length} lines saved to disk]*\n`
930
- })
931
- }
932
-
933
- export function formatMoAResponse({ moaResult, presetName }) {
934
- const parts = []
935
- const pName = presetName || moaResult?.presetName || 'default'
936
- const judge = moaResult?.aggregator || 'unknown'
937
- const refs = moaResult?.references || []
938
-
939
- if (moaResult?.kind === 'questions') {
940
- parts.push('## 🧠 Mixture of Agents — Requirements Clarification')
941
- parts.push(`*Judge (${judge}) and the advisors analyzed the task:*\n`)
942
- parts.push(moaResult.content)
943
- return parts.join('\n')
944
- }
945
-
946
- if (moaResult?.kind === 'failure') {
947
- parts.push('## ⚠️ Mixture of Agents — Execution Failed')
948
- parts.push(moaResult.content)
949
- return parts.join('\n')
950
- }
951
-
952
- const modeBadge = moaResult?.isFastMode
953
- ? '⚡ Fast Mode'
954
- : (moaResult?.isCuratorSynthesis ? `🧠 Curator: ${judge}` : `Judge: ${judge}`)
955
- parts.push(`## 🧠 Mixture of Agents (Preset: ${pName} | ${modeBadge})`)
956
- parts.push('')
957
-
958
- if (moaResult?.isRefinement) {
959
- parts.push('> 🔄 **Mode**: Iterative project refinement (Refinement)')
960
- }
961
-
962
- const hasPromoted = moaResult?.promotedFiles && moaResult.promotedFiles.length > 0
963
- if (hasPromoted) {
964
- parts.push(`> 📦 **Files created in the project**: \`${moaResult.promotedFiles.join('`, `')}\``)
965
- }
966
-
967
- if (moaResult?.recommendedAssembler?.label) {
968
- parts.push(`> 🎯 **Recommended Master Assembler**: Candidate ${moaResult.recommendedAssembler.index} (\`${moaResult.recommendedAssembler.label}\`)`)
969
- }
970
-
971
- if (moaResult?.liveCanvas?.previewUrl) {
972
- parts.push(`> 🎨 **Live Canvas**: [🚀 Открыть ${moaResult.liveCanvas.title || 'превью'} в Live Canvas](${moaResult.liveCanvas.previewUrl}) | [↗ Открыть в новой вкладке](${moaResult.liveCanvas.previewUrl})`)
973
- }
974
-
975
- // Cost tracking card
976
- if (moaResult?.usage) {
977
- const u = moaResult.usage
978
- const costStr = u.totalCostUsd > 0 ? `~\$${u.totalCostUsd.toFixed(4)}` : 'Free'
979
- const tokStr = u.totalTokens >= 1000 ? `${(u.totalTokens / 1000).toFixed(1)}k` : `${u.totalTokens}`
980
- parts.push(`> 💰 **Run cost**: ${costStr} (${tokStr} tokens total)`)
981
- }
982
-
983
- parts.push('')
984
-
985
- if (!moaResult?.isFastMode) {
986
- parts.push(`### ⚖️ Judge verdict and final synthesis (Synthesis: ${judge})`)
987
- parts.push('')
988
- const cleanJudgeContent = hasPromoted
989
- ? stripOrSummarizeCode(moaResult?.content || '')
990
- : (moaResult?.content || '(нет ответа)')
991
- parts.push(cleanJudgeContent)
992
- parts.push('')
993
- }
994
-
995
- if (Array.isArray(refs) && refs.length > 0) {
996
- const title = moaResult?.isFastMode ? '### 🚀 Candidate generation result:' : `### 👥 Advisor responses (${refs.length}):`
997
- parts.push(title)
998
- parts.push('')
999
- refs.forEach((ref, i) => {
1000
- const statusIcon = ref.ok ? '✅' : '⚠️'
1001
- const fileBadge = ref.files?.length ? ` (${ref.files.length} файл(ов))` : ''
1002
- const costBadge = ref.costUsd > 0 ? ` [~\$${ref.costUsd.toFixed(4)}]` : ''
1003
- parts.push(`#### ${statusIcon} Model ${i + 1}: ${ref.label}${fileBadge}${costBadge}`)
1004
- parts.push('')
1005
- parts.push(stripOrSummarizeCode(ref.text))
1006
- parts.push('')
1007
- parts.push('---')
1008
- parts.push('')
1009
- })
1010
- }
1011
-
1012
- return parts.join('\n')
1013
- }
1014
-
1015
675
  export async function* streamMoATurn({ targetPreset, userPrompt, messages, callLlm, cwd, prices, historyFilePath, liveCanvas }, options = {}) {
1016
676
  const signal = options?.signal
1017
677
  if (signal?.aborted) return
@@ -1116,9 +776,6 @@ export async function* streamMoATurn({ targetPreset, userPrompt, messages, callL
1116
776
  }
1117
777
  yield { type: 'finish', reason: { kind: 'stop' } }
1118
778
  } finally {
1119
- if (signal?.aborted) {
1120
- await cleanMoaWorkspaces(cwd)
1121
- }
779
+ // cleanup
1122
780
  }
1123
781
  }
1124
-