@goodandready/dsh-moa 0.2.11 → 0.2.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -23,7 +23,9 @@ import {
23
23
  getMoaHistory,
24
24
  getMoaLeaderboard,
25
25
  getMoaRunById,
26
+ exportMoaHistory,
26
27
  } from './history.js'
28
+ import { promoteCandidateWorkspace } from './file-workspace.js'
27
29
 
28
30
  import { createLiveCanvasClient } from './live-canvas.js'
29
31
 
@@ -41,6 +43,7 @@ export const PriceRow = z.object({
41
43
  export const ModelSlotSchema = z.object({
42
44
  provider: z.string().default('opencode-go'),
43
45
  model: z.string().default('deepseek-v4-flash'),
46
+ role_persona: z.string().default(''),
44
47
  })
45
48
 
46
49
  export const PresetSchema = z.object({
@@ -58,6 +61,9 @@ export const PresetSchema = z.object({
58
61
  aggregator_temperature: z.number().default(0.4),
59
62
  reference_timeout_sec: z.number().default(60),
60
63
  aggregator_timeout_sec: z.number().default(180),
64
+ blind_evaluation: z.boolean().default(false),
65
+ peer_critique_enabled: z.boolean().default(false),
66
+ allow_candidate_override: z.boolean().default(false),
61
67
  max_tokens: z.number().default(4096),
62
68
  judge_criteria: z.string().default(''),
63
69
  })
@@ -420,7 +426,7 @@ export function apply(ctx, config) {
420
426
  })
421
427
  }, 'dsh-moa: history route')
422
428
 
423
- // Route: /dsh-moa/leaderboard (#11)
429
+ // Route: /dsh-moa/leaderboard
424
430
  ctx.effect(() => {
425
431
  return ctx.webServer.register({
426
432
  kind: 'exact',
@@ -431,7 +437,21 @@ export function apply(ctx, config) {
431
437
  return
432
438
  }
433
439
  try {
434
- const data = getMoaLeaderboard()
440
+ const url = new URL(req.url || '', 'http://127.0.0.1')
441
+ const preset = url.searchParams.get('preset') || null
442
+ const format = url.searchParams.get('format') || 'json'
443
+
444
+ if (format === 'csv') {
445
+ const csv = exportMoaHistory(undefined, { format: 'csv', preset })
446
+ res.writeHead(200, {
447
+ 'Content-Type': 'text/csv; charset=utf-8',
448
+ 'Content-Disposition': 'attachment; filename="moa-leaderboard.csv"',
449
+ })
450
+ res.end(csv)
451
+ return
452
+ }
453
+
454
+ const data = getMoaLeaderboard(undefined, { preset })
435
455
  writeJson(res, 200, { ok: true, ...data })
436
456
  } catch (err) {
437
457
  writeJson(res, 500, { ok: false, error: err?.message || String(err) })
@@ -440,6 +460,71 @@ export function apply(ctx, config) {
440
460
  })
441
461
  }, 'dsh-moa: leaderboard route')
442
462
 
463
+ // Route: /dsh-moa/promote (Candidate Override Action)
464
+ ctx.effect(() => {
465
+ return ctx.webServer.register({
466
+ kind: 'exact',
467
+ path: '/dsh-moa/promote',
468
+ handler: async (req, res) => {
469
+ if (req.method !== 'POST') {
470
+ writeJson(res, 405, { ok: false, error: 'POST required' })
471
+ return
472
+ }
473
+ try {
474
+ const raw = await readBody(req)
475
+ const body = JSON.parse(raw || '{}')
476
+ const candidateIndex = parseInt(body?.candidateIndex, 10)
477
+ if (isNaN(candidateIndex) || candidateIndex < 1) {
478
+ writeJson(res, 400, { ok: false, error: 'Valid candidateIndex required (>= 1)' })
479
+ return
480
+ }
481
+ const cwd = body?.cwd || process.cwd()
482
+ const promotedFiles = await promoteCandidateWorkspace(cwd, candidateIndex, { keepMoa: true })
483
+ writeJson(res, 200, { ok: true, candidateIndex, promotedFiles })
484
+ } catch (err) {
485
+ writeJson(res, 500, { ok: false, error: err?.message || String(err) })
486
+ }
487
+ },
488
+ })
489
+ }, 'dsh-moa: promote candidate route')
490
+
491
+ // Route: /dsh-moa/history/export
492
+ ctx.effect(() => {
493
+ return ctx.webServer.register({
494
+ kind: 'exact',
495
+ path: '/dsh-moa/history/export',
496
+ handler: async (req, res) => {
497
+ if (req.method !== 'GET') {
498
+ writeJson(res, 405, { ok: false, error: 'GET required' })
499
+ return
500
+ }
501
+ try {
502
+ const url = new URL(req.url || '', 'http://127.0.0.1')
503
+ const preset = url.searchParams.get('preset') || null
504
+ const format = url.searchParams.get('format') || 'json'
505
+ const data = exportMoaHistory(undefined, { format, preset })
506
+
507
+ if (format === 'csv') {
508
+ res.writeHead(200, {
509
+ 'Content-Type': 'text/csv; charset=utf-8',
510
+ 'Content-Disposition': 'attachment; filename="moa-history.csv"',
511
+ })
512
+ res.end(data)
513
+ return
514
+ }
515
+
516
+ res.writeHead(200, {
517
+ 'Content-Type': 'application/json; charset=utf-8',
518
+ 'Content-Disposition': 'attachment; filename="moa-history.json"',
519
+ })
520
+ res.end(data)
521
+ } catch (err) {
522
+ writeJson(res, 500, { ok: false, error: err?.message || String(err) })
523
+ }
524
+ },
525
+ })
526
+ }, 'dsh-moa: history export route')
527
+
443
528
  // Route: /dsh-moa/runs/<id> (run replay by id)
444
529
  ctx.effect(() => {
445
530
  return ctx.webServer.register({
package/lib/moa-parser.js CHANGED
@@ -45,6 +45,17 @@ export function parseMoACommand(text, presets = []) {
45
45
  return null
46
46
  }
47
47
 
48
+ // Check for promote subcommand: /moa promote <runId> <candidateIndex> or /moa promote <candidateIndex>
49
+ const promoteMatch = /^\/moa\s+promote\s+([a-zA-Z0-9_\-]+)(?:\s+(\d+))?/i.exec(text.trim())
50
+ if (promoteMatch) {
51
+ const hasTwoArgs = Boolean(promoteMatch[2])
52
+ return {
53
+ isPromote: true,
54
+ runId: hasTwoArgs ? promoteMatch[1] : null,
55
+ candidateIndex: hasTwoArgs ? parseInt(promoteMatch[2], 10) : parseInt(promoteMatch[1], 10),
56
+ }
57
+ }
58
+
48
59
  const remainder = text.slice(4).trim()
49
60
  if (!remainder) {
50
61
  return {
@@ -177,5 +188,18 @@ export function formatMoAResponse({ moaResult, presetName }) {
177
188
  })
178
189
  }
179
190
 
191
+ if (moaResult?.allowCandidateOverride && refs.length > 1) {
192
+ const runId = moaResult?.runId || ''
193
+ parts.push('### 🔄 Candidate Override Actions')
194
+ parts.push('To apply an alternative candidate\'s files instead of the judge\'s pick:')
195
+ refs.forEach((ref, idx) => {
196
+ const isWinner = (idx + 1) === moaResult.winningIndex
197
+ const tag = isWinner ? ' *(Current Judge Pick)*' : ''
198
+ const promoteCmd = runId ? `/moa promote ${runId} ${idx + 1}` : `/moa promote ${idx + 1}`
199
+ parts.push(`- \`${promoteCmd}\` — Candidate ${idx + 1} (${ref.label})${tag}`)
200
+ })
201
+ parts.push('')
202
+ }
203
+
180
204
  return parts.join('\n')
181
205
  }
@@ -33,6 +33,38 @@ Penalize and strictly downgrade candidates exhibiting any of the following flaws
33
33
  8. 🚫 Context Amnesia & Regressions:
34
34
  - Dropping or breaking previously functioning project features while adding new code.`
35
35
 
36
+ export const ROLE_PERSONA_PROMPTS = {
37
+ minimalist: `### 🎯 Specialized Engineering Persona: THE MINIMALIST (Ponytail)
38
+ - Prioritize the standard library and native platform capabilities over third-party dependencies.
39
+ - Zero boilerplate, no premature abstractions, interfaces, or unneeded layers.
40
+ - Write the shortest, cleanest working solution with maximum readability.`,
41
+
42
+ robustness: `### 🎯 Specialized Engineering Persona: ROBUSTNESS & DEFENSIVE DESIGN
43
+ - Emphasize paranoid input validation, boundary checking, and comprehensive error handling.
44
+ - Anticipate network failures, null/undefined edge cases, malformed data, and race conditions.
45
+ - Ensure grace under failure and informative, actionable error reporting.`,
46
+
47
+ performance: `### 🎯 Specialized Engineering Persona: HIGH PERFORMANCE & EFFICIENCY
48
+ - Prioritize asymptotic time and space complexity, minimal memory allocations, and zero redundant work.
49
+ - Optimize hot paths, streaming data processing, and efficient algorithmic structures.
50
+ - Document the Big-O complexity and performance characteristics of your solution.`,
51
+
52
+ tester: `### 🎯 Specialized Engineering Persona: TESTABILITY & VERIFICATION
53
+ - Structure code for maximum testability, modular decoupling, and clear side-effect boundaries.
54
+ - Include comprehensive unit/integration test specifications verifying happy paths and edge cases.
55
+ - Emphasize deterministic behavior and assertions.`,
56
+
57
+ general: '',
58
+ }
59
+
60
+ export const SYNTAX_CORRECTION_DIRECTIVE = `### ⚠️ Syntax Warning & Auto-Fix Directive:
61
+ Some candidate proposals have detected syntax flaws (annotated with [⚠️ Syntax Warning]).
62
+ CRITICAL JUDGE DIRECTIVE: If a candidate with a syntax flaw presents superior architecture, algorithm, or engineering logic compared to other candidates, DO NOT reject them solely for this syntax flaw!
63
+ Instead, you MUST correct the syntax error directly during synthesis/assembly and declare that candidate the winner.`
64
+
65
+ export const LANGUAGE_MIRRORING_DIRECTIVE = `### 🌐 Language Mirroring Requirement
66
+ CRITICAL: You MUST write your entire analysis, reasoning, verdict, explanations and instructions in the EXACT SAME LANGUAGE as the user's prompt (e.g. Russian if the prompt is written in Russian, English if in English). Do NOT switch or translate to English unless explicitly requested by the user.`
67
+
36
68
  /**
37
69
  * Formats a provider + model slot into a readable string key.
38
70
  */
@@ -141,15 +173,50 @@ The user gave the task:
141
173
  The advisors proposed the following decision points and clarifications:
142
174
  ${joined}
143
175
 
176
+ ${LANGUAGE_MIRRORING_DIRECTIVE}
177
+
144
178
  Your task is to synthesize a single, compact, friendly and structured questionnaire (2-4 questions) in the same language as the user's prompt.
145
179
  Each question must offer 2-3 concrete recommended answer options (e.g.: 1. Format: single-file HTML/JS or React? 2. Style: minimalism, iOS or neubrutalism?).
146
180
  At the end, add a note that the user can answer briefly (e.g.: "1, 2, dark theme") or trust the defaults.`
147
181
  }
148
182
 
149
183
  /**
150
- * Builds the curator synthesis prompt with component analysis and model recommendation.
184
+ * Builds the curator synthesis prompt with component analysis, model recommendation, and optional blind review.
151
185
  */
152
- export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '') {
186
+ /**
187
+ * Builds the Round 2 Consilium / Peer Critique prompt for candidate models.
188
+ */
189
+ export function buildPeerCritiquePrompt(userPrompt, myProposal, otherProposals = [], isBlind = true) {
190
+ const othersText = otherProposals
191
+ .map((p, idx) => {
192
+ const label = isBlind ? `Candidate ${idx + 1}` : (p.label || `Candidate ${idx + 1}`)
193
+ return `### ${label} Alternative Proposal:
194
+ ${p.text}`
195
+ })
196
+ .join('\n\n---\n\n')
197
+
198
+ return `You are participating in Round 2 (Consilium / Peer Critique & Solution Refinement) of an MoA ensemble.
199
+
200
+ Original User Request:
201
+ ${userPrompt}
202
+
203
+ Your Initial Solution (Round 1):
204
+ ${myProposal}
205
+
206
+ Alternative Solutions from Peer Candidates:
207
+ ${othersText}
208
+
209
+ Instructions:
210
+ 1. 🔍 Peer Critique: Briefly evaluate the other candidates' solutions. Point out any hidden bugs, race conditions, antipatterns, or edge-case omissions in their logic.
211
+ 2. 🛡️ Defend & Refine: Adopt the strongest architectural ideas or optimizations from competitors to enhance your own solution. Fix any oversights in your own code.
212
+ 3. 🚀 Final Refined Deliverable: Present your complete, production-ready, improved code with full explanations.
213
+
214
+ ${LANGUAGE_MIRRORING_DIRECTIVE}`
215
+ }
216
+
217
+ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
218
+ const isBlind = Boolean(options.blindEvaluation)
219
+
153
220
  const joined = referenceOutputs
154
221
  .map((r, i) => {
155
222
  const fileSummary = (r.files && r.files.length > 0)
@@ -159,7 +226,11 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
159
226
  if (referenceOutputs.length >= 3 && textContent.length > 3000) {
160
227
  textContent = stripOrSummarizeCode(textContent)
161
228
  }
162
- return `Candidate ${i + 1} — ${r.label}:${fileSummary}\n${textContent}`
229
+ const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
230
+ const header = isBlind
231
+ ? `Candidate ${i + 1}:${syntaxNote}${fileSummary}`
232
+ : `Candidate ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
233
+ return `${header}\n${textContent}`
163
234
  })
164
235
  .join('\n\n')
165
236
 
@@ -178,8 +249,12 @@ ${joined}
178
249
 
179
250
  ${ANTIPATTERNS_RUBRIC}
180
251
 
252
+ ${LANGUAGE_MIRRORING_DIRECTIVE}
253
+ ${referenceOutputs.some((r) => r.syntaxWarning) ? `
254
+ ${SYNTAX_CORRECTION_DIRECTIVE}
255
+ ` : ''}
181
256
  Instructions:
182
- Your response MUST be structured into three clear parts (respond in the same language as the user's prompt, e.g. Russian):
257
+ Your response MUST be structured into three clear parts:
183
258
 
184
259
  ### 1. 🔍 Curator Analysis & Component Breakdown
185
260
  - For EACH candidate, provide:
@@ -189,7 +264,7 @@ Your response MUST be structured into three clear parts (respond in the same lan
189
264
 
190
265
  ### 2. 🧩 Assembly Recipe & Recommended Master Assembler
191
266
  - Recommend the best single agent model to assemble and finalize the solution:
192
- RECOMMENDED_ASSEMBLER: <number from 1 to N> (<provider:model>)
267
+ RECOMMENDED_ASSEMBLER: <number from 1 to N> ${isBlind ? '' : '(<provider:model>)'}
193
268
  - State the machine winner index marker for file promotion:
194
269
  WINNER_CANDIDATE_INDEX: <number from 1 to N>
195
270
  - Provide the exact blueprint / instructions for combining the best pieces into a unified deliverable.
@@ -200,13 +275,15 @@ Your response MUST be structured into three clear parts (respond in the same lan
200
275
  }
201
276
 
202
277
  /**
203
- * Builds the judge synthesis prompt for standard or curator mode.
278
+ * Builds the judge synthesis prompt for standard or curator mode with optional blind evaluation.
204
279
  */
205
280
  export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
206
281
  if (options.curatorSynthesis) {
207
- return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria)
282
+ return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, options)
208
283
  }
209
284
 
285
+ const isBlind = Boolean(options.blindEvaluation)
286
+
210
287
  const joined = referenceOutputs
211
288
  .map((r, i) => {
212
289
  const fileSummary = (r.files && r.files.length > 0)
@@ -216,7 +293,11 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
216
293
  if (referenceOutputs.length >= 3 && textContent.length > 3000) {
217
294
  textContent = stripOrSummarizeCode(textContent)
218
295
  }
219
- return `Reference ${i + 1} — ${r.label}:${fileSummary}\n${textContent}`
296
+ const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
297
+ const header = isBlind
298
+ ? `Reference ${i + 1}:${syntaxNote}${fileSummary}`
299
+ : `Reference ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
300
+ return `${header}\n${textContent}`
220
301
  })
221
302
  .join('\n\n')
222
303
 
@@ -234,11 +315,13 @@ ${joined}
234
315
 
235
316
  ${ANTIPATTERNS_RUBRIC}
236
317
 
318
+ ${LANGUAGE_MIRRORING_DIRECTIVE}
319
+ ${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
237
320
  Instructions:
238
- Your response MUST be structured into three clear parts (respond in the same language as the user's prompt, e.g. Russian):
321
+ Your response MUST be structured into three clear parts:
239
322
 
240
323
  ### 1. ⚖️ Judge verdict and comparative analysis
241
- - **Winner**: clearly name the model and candidate number (e.g. "Winner: Candidate 1 (opencode-go:deepseek-v4-flash)" or "Reference 1 (label) is chosen").
324
+ - **Winner**: clearly name the chosen candidate (e.g. "Winner: Candidate 1" or "Reference 1 is chosen").
242
325
  - Always add the machine winner-selection marker:
243
326
  WINNER_CANDIDATE_INDEX: <number from 1 to N>
244
327
  - **Why this choice**: compare code, architecture, strengths, weaknesses and reliability of all candidates in detail.
package/lib/moa-runner.js CHANGED
@@ -11,53 +11,16 @@
11
11
  */
12
12
 
13
13
  import path from 'node:path'
14
- import {
15
- extractFileBlocks,
16
- collectProjectContext,
17
- isRefinementTask,
18
- writeCandidateWorkspace,
19
- promoteCandidateWorkspace,
20
- cleanMoaWorkspaces,
21
- } from './file-workspace.js'
14
+ import crypto from 'node:crypto'
15
+ import { extractFileBlocks, collectProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
22
16
  import { estimateTokenCost } from './pricing.js'
23
17
  import { recordMoaRun, recordMoaRunAsync } from './history.js'
24
- import {
25
- slotLabel,
26
- cleanAdvisoryMessages,
27
- isBroadPromptRequiringQuestions,
28
- buildQuestionSynthesisPrompt,
29
- buildCuratorSynthesisPrompt,
30
- buildSynthesisPrompt,
31
- SYSTEM_ROLE_PROPOSER,
32
- ANTIPATTERNS_RUBRIC,
33
- } from './moa-prompts.js'
34
- import {
35
- parseWinnerIndex,
36
- parseRecommendedAssembler,
37
- parseMoACommand,
38
- stripOrSummarizeCode,
39
- formatMoAResponse,
40
- } from './moa-parser.js'
41
-
42
- // Re-export prompt and parser functions for external consumers / backwards compatibility
43
- export {
44
- slotLabel,
45
- cleanAdvisoryMessages,
46
- isBroadPromptRequiringQuestions,
47
- buildQuestionSynthesisPrompt,
48
- buildCuratorSynthesisPrompt,
49
- buildSynthesisPrompt,
50
- ANTIPATTERNS_RUBRIC,
51
- } from './moa-prompts.js'
52
-
53
- export {
54
- parseWinnerIndex,
55
- parseRecommendedAssembler,
56
- parseMoACommand,
57
- stripOrSummarizeCode,
58
- formatMoAResponse,
59
- } from './moa-parser.js'
18
+ import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
19
+ import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
60
20
 
21
+ // Re-exports for consumers & backward compatibility
22
+ export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
23
+ export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
61
24
  export { estimateTokenCost } from './pricing.js'
62
25
 
63
26
  export const DEFAULT_PROMPT = 'Please propose an optimal, well-structured, production-ready solution with full code and explanations.'
@@ -94,8 +57,6 @@ export async function runReferencesParallel(references, messages, options = {},
94
57
 
95
58
  const advisoryMessages = cleanAdvisoryMessages(messages)
96
59
  const systemPrompt = options.systemPrompt || REFERENCE_SYSTEM_PROMPT
97
- // Deterministic prefix for optimal prompt caching hit rate
98
- const fullMessages = [{ role: 'system', content: systemPrompt }, ...advisoryMessages]
99
60
  const timeoutMs = options.timeoutMs ?? ((options.timeoutSec ?? 60) * 1000)
100
61
  const maxRetries = options.maxRetries ?? 0
101
62
  const quorumEnabled = Boolean(options.quorumEnabled)
@@ -117,6 +78,10 @@ export async function runReferencesParallel(references, messages, options = {},
117
78
 
118
79
  references.forEach((slot, i) => {
119
80
  const label = slotLabel(slot)
81
+ const persona = slot?.role_persona && ROLE_PERSONA_PROMPTS[slot.role_persona]
82
+ ? `\n\n${ROLE_PERSONA_PROMPTS[slot.role_persona]}`
83
+ : ''
84
+ const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
120
85
 
121
86
  const runOne = async () => {
122
87
  try {
@@ -125,7 +90,7 @@ export async function runReferencesParallel(references, messages, options = {},
125
90
  {
126
91
  provider: slot.provider,
127
92
  model: slot.model,
128
- messages: fullMessages,
93
+ messages: slotMessages,
129
94
  temperature: options.temperature ?? 0.6,
130
95
  maxTokens: options.maxTokens ?? 4096,
131
96
  timeoutMs,
@@ -144,7 +109,7 @@ export async function runReferencesParallel(references, messages, options = {},
144
109
  const text = typeof res === 'string' ? res : (res?.content || res?.text || '')
145
110
 
146
111
  const fallbackUsage = (typeof res === 'object' && res?.usage) ? res.usage : {
147
- inputTokens: Math.max(1, Math.round(fullMessages.map((m) => m.content).join('').length / 4)),
112
+ inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
148
113
  outputTokens: Math.max(1, Math.round(text.length / 4)),
149
114
  }
150
115
  const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
@@ -327,6 +292,10 @@ export async function runMoAPipeline({
327
292
  const candidateRetries = 1
328
293
  const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
329
294
  const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
295
+ const isBlindEvaluation = Boolean(preset?.blind_evaluation)
296
+ const isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
297
+ const allowCandidateOverride = Boolean(preset?.allow_candidate_override)
298
+ const runId = crypto.randomUUID()
330
299
 
331
300
  // 2. Collect project context for refinement tasks
332
301
  const isRefinement = isRefinementTask(userPrompt)
@@ -530,8 +499,50 @@ export async function runMoAPipeline({
530
499
  }
531
500
  }
532
501
 
502
+ // 7b. Consilium Round 2 (Peer Critique) if enabled
503
+ if (isPeerCritiqueEnabled && successfulRefs.length > 1) {
504
+ if (typeof onProgress === 'function') {
505
+ onProgress('🤝 *Consilium Round 2: Candidates reviewing peer proposals in parallel...*\n')
506
+ }
507
+ const r2Promises = successfulRefs.map(async (cand) => {
508
+ const opponents = successfulRefs
509
+ .filter((c) => c.index !== cand.index)
510
+ .map((c) => ({ label: isBlindEvaluation ? `Candidate ${c.index}` : c.label, text: c.text }))
511
+ const r2Prompt = buildPeerCritiquePrompt(userPrompt, cand.text, opponents, isBlindEvaluation)
512
+ try {
513
+ const slot = referenceModels[cand.index - 1] || { provider: cand.provider, model: cand.model }
514
+ const res = await callWithTransientRetry(callLlm, {
515
+ provider: slot.provider,
516
+ model: slot.model,
517
+ messages: [{ role: 'user', content: r2Prompt }],
518
+ temperature: refTemp,
519
+ maxTokens,
520
+ timeoutMs: refTimeoutSec * 1000,
521
+ }, candidateRetries)
522
+ const refinedText = typeof res === 'string' ? res : (res?.content || res?.text || cand.text)
523
+ cand.text = refinedText
524
+ const r2Files = extractFileBlocks(refinedText)
525
+ if (r2Files.length > 0) {
526
+ cand.files = r2Files
527
+ cand.syntaxWarning = (verifyFileSyntax(r2Files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
528
+ }
529
+ if (typeof res === 'object' && res?.usage) {
530
+ cand.usage.inputTokens = (cand.usage.inputTokens || 0) + (res.usage.inputTokens || 0)
531
+ cand.usage.outputTokens = (cand.usage.outputTokens || 0) + (res.usage.outputTokens || 0)
532
+ cand.usage.totalTokens = (cand.usage.totalTokens || 0) + (res.usage.totalTokens || 0)
533
+ const extraCost = estimateTokenCost(slot, res.usage, prices)
534
+ cand.costUsd = Number(((cand.costUsd || 0) + extraCost.costUsd).toFixed(5))
535
+ }
536
+ } catch {}
537
+ })
538
+ await Promise.allSettled(r2Promises)
539
+ }
540
+
533
541
  // 8. Synthesis phase via primary judge or fallback chain
534
- const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, { curatorSynthesis: isCuratorSynthesis })
542
+ const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
543
+ curatorSynthesis: isCuratorSynthesis,
544
+ blindEvaluation: isBlindEvaluation,
545
+ })
535
546
  let synthesizedText = ''
536
547
  let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
537
548
  let chosenJudge = primaryJudge
@@ -600,15 +611,16 @@ export async function runMoAPipeline({
600
611
  const synthesizedFiles = extractFileBlocks(synthesizedText)
601
612
  let promotedFiles = []
602
613
 
614
+ const keepWorkspaces = allowCandidateOverride
603
615
  if (synthesizedFiles.length > 0) {
604
616
  await writeCandidateWorkspace(cwd, 'curator-synthesis', synthesizedFiles)
605
- promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis')
617
+ promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis', { keepMoa: keepWorkspaces })
606
618
  } else if (referenceOutputs[winningIndex - 1]?.files?.length > 0) {
607
- promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex)
619
+ promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex, { keepMoa: keepWorkspaces })
608
620
  } else if (successfulRefs[0]?.files?.length > 0) {
609
621
  const fallbackIdx = successfulRefs[0].index
610
- promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx)
611
- } else {
622
+ promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx, { keepMoa: keepWorkspaces })
623
+ } else if (!keepWorkspaces) {
612
624
  await cleanMoaWorkspaces(cwd)
613
625
  }
614
626
 
@@ -625,6 +637,7 @@ export async function runMoAPipeline({
625
637
  // Record run to history
626
638
  try {
627
639
  recordMoaRun({
640
+ id: runId,
628
641
  prompt: userPrompt,
629
642
  preset: preset?.name || 'default',
630
643
  isRefinement,
@@ -654,6 +667,8 @@ export async function runMoAPipeline({
654
667
  winningIndex,
655
668
  winnerModel,
656
669
  promotedFiles,
670
+ runId,
671
+ allowCandidateOverride,
657
672
  ...(livePreview ? { liveCanvas: livePreview } : {}),
658
673
  usage: {
659
674
  totalTokens,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@goodandready/dsh-moa",
3
- "version": "0.2.11",
3
+ "version": "0.2.13",
4
4
  "description": "Mixture of Agents (MoA) plugin for DeepSeek Harness with /moa slash command",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",