@goodandready/dsh-moa 0.2.11 → 0.2.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +49 -2
- package/README.ru.md +287 -0
- package/README.zh.md +243 -0
- package/docs/README.ru.md +48 -1
- package/docs/README.zh.md +11 -1
- package/docs/design/DESIGN.md +3 -0
- package/docs/plans/59-power-pack-plan.md +73 -0
- package/lib/client.js +225 -0
- package/lib/file-workspace.js +46 -0
- package/lib/history.js +56 -2
- package/lib/index.js +87 -2
- package/lib/moa-parser.js +24 -0
- package/lib/moa-prompts.js +93 -10
- package/lib/moa-runner.js +68 -53
- package/package.json +1 -1
package/lib/index.js
CHANGED
|
@@ -23,7 +23,9 @@ import {
|
|
|
23
23
|
getMoaHistory,
|
|
24
24
|
getMoaLeaderboard,
|
|
25
25
|
getMoaRunById,
|
|
26
|
+
exportMoaHistory,
|
|
26
27
|
} from './history.js'
|
|
28
|
+
import { promoteCandidateWorkspace } from './file-workspace.js'
|
|
27
29
|
|
|
28
30
|
import { createLiveCanvasClient } from './live-canvas.js'
|
|
29
31
|
|
|
@@ -41,6 +43,7 @@ export const PriceRow = z.object({
|
|
|
41
43
|
export const ModelSlotSchema = z.object({
|
|
42
44
|
provider: z.string().default('opencode-go'),
|
|
43
45
|
model: z.string().default('deepseek-v4-flash'),
|
|
46
|
+
role_persona: z.string().default(''),
|
|
44
47
|
})
|
|
45
48
|
|
|
46
49
|
export const PresetSchema = z.object({
|
|
@@ -58,6 +61,9 @@ export const PresetSchema = z.object({
|
|
|
58
61
|
aggregator_temperature: z.number().default(0.4),
|
|
59
62
|
reference_timeout_sec: z.number().default(60),
|
|
60
63
|
aggregator_timeout_sec: z.number().default(180),
|
|
64
|
+
blind_evaluation: z.boolean().default(false),
|
|
65
|
+
peer_critique_enabled: z.boolean().default(false),
|
|
66
|
+
allow_candidate_override: z.boolean().default(false),
|
|
61
67
|
max_tokens: z.number().default(4096),
|
|
62
68
|
judge_criteria: z.string().default(''),
|
|
63
69
|
})
|
|
@@ -420,7 +426,7 @@ export function apply(ctx, config) {
|
|
|
420
426
|
})
|
|
421
427
|
}, 'dsh-moa: history route')
|
|
422
428
|
|
|
423
|
-
// Route: /dsh-moa/leaderboard
|
|
429
|
+
// Route: /dsh-moa/leaderboard
|
|
424
430
|
ctx.effect(() => {
|
|
425
431
|
return ctx.webServer.register({
|
|
426
432
|
kind: 'exact',
|
|
@@ -431,7 +437,21 @@ export function apply(ctx, config) {
|
|
|
431
437
|
return
|
|
432
438
|
}
|
|
433
439
|
try {
|
|
434
|
-
const
|
|
440
|
+
const url = new URL(req.url || '', 'http://127.0.0.1')
|
|
441
|
+
const preset = url.searchParams.get('preset') || null
|
|
442
|
+
const format = url.searchParams.get('format') || 'json'
|
|
443
|
+
|
|
444
|
+
if (format === 'csv') {
|
|
445
|
+
const csv = exportMoaHistory(undefined, { format: 'csv', preset })
|
|
446
|
+
res.writeHead(200, {
|
|
447
|
+
'Content-Type': 'text/csv; charset=utf-8',
|
|
448
|
+
'Content-Disposition': 'attachment; filename="moa-leaderboard.csv"',
|
|
449
|
+
})
|
|
450
|
+
res.end(csv)
|
|
451
|
+
return
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
const data = getMoaLeaderboard(undefined, { preset })
|
|
435
455
|
writeJson(res, 200, { ok: true, ...data })
|
|
436
456
|
} catch (err) {
|
|
437
457
|
writeJson(res, 500, { ok: false, error: err?.message || String(err) })
|
|
@@ -440,6 +460,71 @@ export function apply(ctx, config) {
|
|
|
440
460
|
})
|
|
441
461
|
}, 'dsh-moa: leaderboard route')
|
|
442
462
|
|
|
463
|
+
// Route: /dsh-moa/promote (Candidate Override Action)
|
|
464
|
+
ctx.effect(() => {
|
|
465
|
+
return ctx.webServer.register({
|
|
466
|
+
kind: 'exact',
|
|
467
|
+
path: '/dsh-moa/promote',
|
|
468
|
+
handler: async (req, res) => {
|
|
469
|
+
if (req.method !== 'POST') {
|
|
470
|
+
writeJson(res, 405, { ok: false, error: 'POST required' })
|
|
471
|
+
return
|
|
472
|
+
}
|
|
473
|
+
try {
|
|
474
|
+
const raw = await readBody(req)
|
|
475
|
+
const body = JSON.parse(raw || '{}')
|
|
476
|
+
const candidateIndex = parseInt(body?.candidateIndex, 10)
|
|
477
|
+
if (isNaN(candidateIndex) || candidateIndex < 1) {
|
|
478
|
+
writeJson(res, 400, { ok: false, error: 'Valid candidateIndex required (>= 1)' })
|
|
479
|
+
return
|
|
480
|
+
}
|
|
481
|
+
const cwd = body?.cwd || process.cwd()
|
|
482
|
+
const promotedFiles = await promoteCandidateWorkspace(cwd, candidateIndex, { keepMoa: true })
|
|
483
|
+
writeJson(res, 200, { ok: true, candidateIndex, promotedFiles })
|
|
484
|
+
} catch (err) {
|
|
485
|
+
writeJson(res, 500, { ok: false, error: err?.message || String(err) })
|
|
486
|
+
}
|
|
487
|
+
},
|
|
488
|
+
})
|
|
489
|
+
}, 'dsh-moa: promote candidate route')
|
|
490
|
+
|
|
491
|
+
// Route: /dsh-moa/history/export
|
|
492
|
+
ctx.effect(() => {
|
|
493
|
+
return ctx.webServer.register({
|
|
494
|
+
kind: 'exact',
|
|
495
|
+
path: '/dsh-moa/history/export',
|
|
496
|
+
handler: async (req, res) => {
|
|
497
|
+
if (req.method !== 'GET') {
|
|
498
|
+
writeJson(res, 405, { ok: false, error: 'GET required' })
|
|
499
|
+
return
|
|
500
|
+
}
|
|
501
|
+
try {
|
|
502
|
+
const url = new URL(req.url || '', 'http://127.0.0.1')
|
|
503
|
+
const preset = url.searchParams.get('preset') || null
|
|
504
|
+
const format = url.searchParams.get('format') || 'json'
|
|
505
|
+
const data = exportMoaHistory(undefined, { format, preset })
|
|
506
|
+
|
|
507
|
+
if (format === 'csv') {
|
|
508
|
+
res.writeHead(200, {
|
|
509
|
+
'Content-Type': 'text/csv; charset=utf-8',
|
|
510
|
+
'Content-Disposition': 'attachment; filename="moa-history.csv"',
|
|
511
|
+
})
|
|
512
|
+
res.end(data)
|
|
513
|
+
return
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
res.writeHead(200, {
|
|
517
|
+
'Content-Type': 'application/json; charset=utf-8',
|
|
518
|
+
'Content-Disposition': 'attachment; filename="moa-history.json"',
|
|
519
|
+
})
|
|
520
|
+
res.end(data)
|
|
521
|
+
} catch (err) {
|
|
522
|
+
writeJson(res, 500, { ok: false, error: err?.message || String(err) })
|
|
523
|
+
}
|
|
524
|
+
},
|
|
525
|
+
})
|
|
526
|
+
}, 'dsh-moa: history export route')
|
|
527
|
+
|
|
443
528
|
// Route: /dsh-moa/runs/<id> (run replay by id)
|
|
444
529
|
ctx.effect(() => {
|
|
445
530
|
return ctx.webServer.register({
|
package/lib/moa-parser.js
CHANGED
|
@@ -45,6 +45,17 @@ export function parseMoACommand(text, presets = []) {
|
|
|
45
45
|
return null
|
|
46
46
|
}
|
|
47
47
|
|
|
48
|
+
// Check for promote subcommand: /moa promote <runId> <candidateIndex> or /moa promote <candidateIndex>
|
|
49
|
+
const promoteMatch = /^\/moa\s+promote\s+([a-zA-Z0-9_\-]+)(?:\s+(\d+))?/i.exec(text.trim())
|
|
50
|
+
if (promoteMatch) {
|
|
51
|
+
const hasTwoArgs = Boolean(promoteMatch[2])
|
|
52
|
+
return {
|
|
53
|
+
isPromote: true,
|
|
54
|
+
runId: hasTwoArgs ? promoteMatch[1] : null,
|
|
55
|
+
candidateIndex: hasTwoArgs ? parseInt(promoteMatch[2], 10) : parseInt(promoteMatch[1], 10),
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
48
59
|
const remainder = text.slice(4).trim()
|
|
49
60
|
if (!remainder) {
|
|
50
61
|
return {
|
|
@@ -177,5 +188,18 @@ export function formatMoAResponse({ moaResult, presetName }) {
|
|
|
177
188
|
})
|
|
178
189
|
}
|
|
179
190
|
|
|
191
|
+
if (moaResult?.allowCandidateOverride && refs.length > 1) {
|
|
192
|
+
const runId = moaResult?.runId || ''
|
|
193
|
+
parts.push('### 🔄 Candidate Override Actions')
|
|
194
|
+
parts.push('To apply an alternative candidate\'s files instead of the judge\'s pick:')
|
|
195
|
+
refs.forEach((ref, idx) => {
|
|
196
|
+
const isWinner = (idx + 1) === moaResult.winningIndex
|
|
197
|
+
const tag = isWinner ? ' *(Current Judge Pick)*' : ''
|
|
198
|
+
const promoteCmd = runId ? `/moa promote ${runId} ${idx + 1}` : `/moa promote ${idx + 1}`
|
|
199
|
+
parts.push(`- \`${promoteCmd}\` — Candidate ${idx + 1} (${ref.label})${tag}`)
|
|
200
|
+
})
|
|
201
|
+
parts.push('')
|
|
202
|
+
}
|
|
203
|
+
|
|
180
204
|
return parts.join('\n')
|
|
181
205
|
}
|
package/lib/moa-prompts.js
CHANGED
|
@@ -33,6 +33,38 @@ Penalize and strictly downgrade candidates exhibiting any of the following flaws
|
|
|
33
33
|
8. 🚫 Context Amnesia & Regressions:
|
|
34
34
|
- Dropping or breaking previously functioning project features while adding new code.`
|
|
35
35
|
|
|
36
|
+
export const ROLE_PERSONA_PROMPTS = {
|
|
37
|
+
minimalist: `### 🎯 Specialized Engineering Persona: THE MINIMALIST (Ponytail)
|
|
38
|
+
- Prioritize the standard library and native platform capabilities over third-party dependencies.
|
|
39
|
+
- Zero boilerplate, no premature abstractions, interfaces, or unneeded layers.
|
|
40
|
+
- Write the shortest, cleanest working solution with maximum readability.`,
|
|
41
|
+
|
|
42
|
+
robustness: `### 🎯 Specialized Engineering Persona: ROBUSTNESS & DEFENSIVE DESIGN
|
|
43
|
+
- Emphasize paranoid input validation, boundary checking, and comprehensive error handling.
|
|
44
|
+
- Anticipate network failures, null/undefined edge cases, malformed data, and race conditions.
|
|
45
|
+
- Ensure grace under failure and informative, actionable error reporting.`,
|
|
46
|
+
|
|
47
|
+
performance: `### 🎯 Specialized Engineering Persona: HIGH PERFORMANCE & EFFICIENCY
|
|
48
|
+
- Prioritize asymptotic time and space complexity, minimal memory allocations, and zero redundant work.
|
|
49
|
+
- Optimize hot paths, streaming data processing, and efficient algorithmic structures.
|
|
50
|
+
- Document the Big-O complexity and performance characteristics of your solution.`,
|
|
51
|
+
|
|
52
|
+
tester: `### 🎯 Specialized Engineering Persona: TESTABILITY & VERIFICATION
|
|
53
|
+
- Structure code for maximum testability, modular decoupling, and clear side-effect boundaries.
|
|
54
|
+
- Include comprehensive unit/integration test specifications verifying happy paths and edge cases.
|
|
55
|
+
- Emphasize deterministic behavior and assertions.`,
|
|
56
|
+
|
|
57
|
+
general: '',
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export const SYNTAX_CORRECTION_DIRECTIVE = `### ⚠️ Syntax Warning & Auto-Fix Directive:
|
|
61
|
+
Some candidate proposals have detected syntax flaws (annotated with [⚠️ Syntax Warning]).
|
|
62
|
+
CRITICAL JUDGE DIRECTIVE: If a candidate with a syntax flaw presents superior architecture, algorithm, or engineering logic compared to other candidates, DO NOT reject them solely for this syntax flaw!
|
|
63
|
+
Instead, you MUST correct the syntax error directly during synthesis/assembly and declare that candidate the winner.`
|
|
64
|
+
|
|
65
|
+
export const LANGUAGE_MIRRORING_DIRECTIVE = `### 🌐 Language Mirroring Requirement
|
|
66
|
+
CRITICAL: You MUST write your entire analysis, reasoning, verdict, explanations and instructions in the EXACT SAME LANGUAGE as the user's prompt (e.g. Russian if the prompt is written in Russian, English if in English). Do NOT switch or translate to English unless explicitly requested by the user.`
|
|
67
|
+
|
|
36
68
|
/**
|
|
37
69
|
* Formats a provider + model slot into a readable string key.
|
|
38
70
|
*/
|
|
@@ -141,15 +173,50 @@ The user gave the task:
|
|
|
141
173
|
The advisors proposed the following decision points and clarifications:
|
|
142
174
|
${joined}
|
|
143
175
|
|
|
176
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
177
|
+
|
|
144
178
|
Your task is to synthesize a single, compact, friendly and structured questionnaire (2-4 questions) in the same language as the user's prompt.
|
|
145
179
|
Each question must offer 2-3 concrete recommended answer options (e.g.: 1. Format: single-file HTML/JS or React? 2. Style: minimalism, iOS or neubrutalism?).
|
|
146
180
|
At the end, add a note that the user can answer briefly (e.g.: "1, 2, dark theme") or trust the defaults.`
|
|
147
181
|
}
|
|
148
182
|
|
|
149
183
|
/**
|
|
150
|
-
* Builds the curator synthesis prompt with component analysis and
|
|
184
|
+
* Builds the curator synthesis prompt with component analysis, model recommendation, and optional blind review.
|
|
151
185
|
*/
|
|
152
|
-
|
|
186
|
+
/**
|
|
187
|
+
* Builds the Round 2 Consilium / Peer Critique prompt for candidate models.
|
|
188
|
+
*/
|
|
189
|
+
export function buildPeerCritiquePrompt(userPrompt, myProposal, otherProposals = [], isBlind = true) {
|
|
190
|
+
const othersText = otherProposals
|
|
191
|
+
.map((p, idx) => {
|
|
192
|
+
const label = isBlind ? `Candidate ${idx + 1}` : (p.label || `Candidate ${idx + 1}`)
|
|
193
|
+
return `### ${label} Alternative Proposal:
|
|
194
|
+
${p.text}`
|
|
195
|
+
})
|
|
196
|
+
.join('\n\n---\n\n')
|
|
197
|
+
|
|
198
|
+
return `You are participating in Round 2 (Consilium / Peer Critique & Solution Refinement) of an MoA ensemble.
|
|
199
|
+
|
|
200
|
+
Original User Request:
|
|
201
|
+
${userPrompt}
|
|
202
|
+
|
|
203
|
+
Your Initial Solution (Round 1):
|
|
204
|
+
${myProposal}
|
|
205
|
+
|
|
206
|
+
Alternative Solutions from Peer Candidates:
|
|
207
|
+
${othersText}
|
|
208
|
+
|
|
209
|
+
Instructions:
|
|
210
|
+
1. 🔍 Peer Critique: Briefly evaluate the other candidates' solutions. Point out any hidden bugs, race conditions, antipatterns, or edge-case omissions in their logic.
|
|
211
|
+
2. 🛡️ Defend & Refine: Adopt the strongest architectural ideas or optimizations from competitors to enhance your own solution. Fix any oversights in your own code.
|
|
212
|
+
3. 🚀 Final Refined Deliverable: Present your complete, production-ready, improved code with full explanations.
|
|
213
|
+
|
|
214
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}`
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
|
|
218
|
+
const isBlind = Boolean(options.blindEvaluation)
|
|
219
|
+
|
|
153
220
|
const joined = referenceOutputs
|
|
154
221
|
.map((r, i) => {
|
|
155
222
|
const fileSummary = (r.files && r.files.length > 0)
|
|
@@ -159,7 +226,11 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
|
|
|
159
226
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
160
227
|
textContent = stripOrSummarizeCode(textContent)
|
|
161
228
|
}
|
|
162
|
-
|
|
229
|
+
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
230
|
+
const header = isBlind
|
|
231
|
+
? `Candidate ${i + 1}:${syntaxNote}${fileSummary}`
|
|
232
|
+
: `Candidate ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
|
|
233
|
+
return `${header}\n${textContent}`
|
|
163
234
|
})
|
|
164
235
|
.join('\n\n')
|
|
165
236
|
|
|
@@ -178,8 +249,12 @@ ${joined}
|
|
|
178
249
|
|
|
179
250
|
${ANTIPATTERNS_RUBRIC}
|
|
180
251
|
|
|
252
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
253
|
+
${referenceOutputs.some((r) => r.syntaxWarning) ? `
|
|
254
|
+
${SYNTAX_CORRECTION_DIRECTIVE}
|
|
255
|
+
` : ''}
|
|
181
256
|
Instructions:
|
|
182
|
-
Your response MUST be structured into three clear parts
|
|
257
|
+
Your response MUST be structured into three clear parts:
|
|
183
258
|
|
|
184
259
|
### 1. 🔍 Curator Analysis & Component Breakdown
|
|
185
260
|
- For EACH candidate, provide:
|
|
@@ -189,7 +264,7 @@ Your response MUST be structured into three clear parts (respond in the same lan
|
|
|
189
264
|
|
|
190
265
|
### 2. 🧩 Assembly Recipe & Recommended Master Assembler
|
|
191
266
|
- Recommend the best single agent model to assemble and finalize the solution:
|
|
192
|
-
RECOMMENDED_ASSEMBLER: <number from 1 to N> (<provider:model>)
|
|
267
|
+
RECOMMENDED_ASSEMBLER: <number from 1 to N> ${isBlind ? '' : '(<provider:model>)'}
|
|
193
268
|
- State the machine winner index marker for file promotion:
|
|
194
269
|
WINNER_CANDIDATE_INDEX: <number from 1 to N>
|
|
195
270
|
- Provide the exact blueprint / instructions for combining the best pieces into a unified deliverable.
|
|
@@ -200,13 +275,15 @@ Your response MUST be structured into three clear parts (respond in the same lan
|
|
|
200
275
|
}
|
|
201
276
|
|
|
202
277
|
/**
|
|
203
|
-
* Builds the judge synthesis prompt for standard or curator mode.
|
|
278
|
+
* Builds the judge synthesis prompt for standard or curator mode with optional blind evaluation.
|
|
204
279
|
*/
|
|
205
280
|
export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
|
|
206
281
|
if (options.curatorSynthesis) {
|
|
207
|
-
return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria)
|
|
282
|
+
return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, options)
|
|
208
283
|
}
|
|
209
284
|
|
|
285
|
+
const isBlind = Boolean(options.blindEvaluation)
|
|
286
|
+
|
|
210
287
|
const joined = referenceOutputs
|
|
211
288
|
.map((r, i) => {
|
|
212
289
|
const fileSummary = (r.files && r.files.length > 0)
|
|
@@ -216,7 +293,11 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
|
|
|
216
293
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
217
294
|
textContent = stripOrSummarizeCode(textContent)
|
|
218
295
|
}
|
|
219
|
-
|
|
296
|
+
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
297
|
+
const header = isBlind
|
|
298
|
+
? `Reference ${i + 1}:${syntaxNote}${fileSummary}`
|
|
299
|
+
: `Reference ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
|
|
300
|
+
return `${header}\n${textContent}`
|
|
220
301
|
})
|
|
221
302
|
.join('\n\n')
|
|
222
303
|
|
|
@@ -234,11 +315,13 @@ ${joined}
|
|
|
234
315
|
|
|
235
316
|
${ANTIPATTERNS_RUBRIC}
|
|
236
317
|
|
|
318
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
319
|
+
${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
|
|
237
320
|
Instructions:
|
|
238
|
-
Your response MUST be structured into three clear parts
|
|
321
|
+
Your response MUST be structured into three clear parts:
|
|
239
322
|
|
|
240
323
|
### 1. ⚖️ Judge verdict and comparative analysis
|
|
241
|
-
- **Winner**: clearly name the
|
|
324
|
+
- **Winner**: clearly name the chosen candidate (e.g. "Winner: Candidate 1" or "Reference 1 is chosen").
|
|
242
325
|
- Always add the machine winner-selection marker:
|
|
243
326
|
WINNER_CANDIDATE_INDEX: <number from 1 to N>
|
|
244
327
|
- **Why this choice**: compare code, architecture, strengths, weaknesses and reliability of all candidates in detail.
|
package/lib/moa-runner.js
CHANGED
|
@@ -11,53 +11,16 @@
|
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
13
|
import path from 'node:path'
|
|
14
|
-
import
|
|
15
|
-
|
|
16
|
-
collectProjectContext,
|
|
17
|
-
isRefinementTask,
|
|
18
|
-
writeCandidateWorkspace,
|
|
19
|
-
promoteCandidateWorkspace,
|
|
20
|
-
cleanMoaWorkspaces,
|
|
21
|
-
} from './file-workspace.js'
|
|
14
|
+
import crypto from 'node:crypto'
|
|
15
|
+
import { extractFileBlocks, collectProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
|
|
22
16
|
import { estimateTokenCost } from './pricing.js'
|
|
23
17
|
import { recordMoaRun, recordMoaRunAsync } from './history.js'
|
|
24
|
-
import {
|
|
25
|
-
|
|
26
|
-
cleanAdvisoryMessages,
|
|
27
|
-
isBroadPromptRequiringQuestions,
|
|
28
|
-
buildQuestionSynthesisPrompt,
|
|
29
|
-
buildCuratorSynthesisPrompt,
|
|
30
|
-
buildSynthesisPrompt,
|
|
31
|
-
SYSTEM_ROLE_PROPOSER,
|
|
32
|
-
ANTIPATTERNS_RUBRIC,
|
|
33
|
-
} from './moa-prompts.js'
|
|
34
|
-
import {
|
|
35
|
-
parseWinnerIndex,
|
|
36
|
-
parseRecommendedAssembler,
|
|
37
|
-
parseMoACommand,
|
|
38
|
-
stripOrSummarizeCode,
|
|
39
|
-
formatMoAResponse,
|
|
40
|
-
} from './moa-parser.js'
|
|
41
|
-
|
|
42
|
-
// Re-export prompt and parser functions for external consumers / backwards compatibility
|
|
43
|
-
export {
|
|
44
|
-
slotLabel,
|
|
45
|
-
cleanAdvisoryMessages,
|
|
46
|
-
isBroadPromptRequiringQuestions,
|
|
47
|
-
buildQuestionSynthesisPrompt,
|
|
48
|
-
buildCuratorSynthesisPrompt,
|
|
49
|
-
buildSynthesisPrompt,
|
|
50
|
-
ANTIPATTERNS_RUBRIC,
|
|
51
|
-
} from './moa-prompts.js'
|
|
52
|
-
|
|
53
|
-
export {
|
|
54
|
-
parseWinnerIndex,
|
|
55
|
-
parseRecommendedAssembler,
|
|
56
|
-
parseMoACommand,
|
|
57
|
-
stripOrSummarizeCode,
|
|
58
|
-
formatMoAResponse,
|
|
59
|
-
} from './moa-parser.js'
|
|
18
|
+
import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
19
|
+
import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
60
20
|
|
|
21
|
+
// Re-exports for consumers & backward compatibility
|
|
22
|
+
export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
23
|
+
export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
61
24
|
export { estimateTokenCost } from './pricing.js'
|
|
62
25
|
|
|
63
26
|
export const DEFAULT_PROMPT = 'Please propose an optimal, well-structured, production-ready solution with full code and explanations.'
|
|
@@ -94,8 +57,6 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
94
57
|
|
|
95
58
|
const advisoryMessages = cleanAdvisoryMessages(messages)
|
|
96
59
|
const systemPrompt = options.systemPrompt || REFERENCE_SYSTEM_PROMPT
|
|
97
|
-
// Deterministic prefix for optimal prompt caching hit rate
|
|
98
|
-
const fullMessages = [{ role: 'system', content: systemPrompt }, ...advisoryMessages]
|
|
99
60
|
const timeoutMs = options.timeoutMs ?? ((options.timeoutSec ?? 60) * 1000)
|
|
100
61
|
const maxRetries = options.maxRetries ?? 0
|
|
101
62
|
const quorumEnabled = Boolean(options.quorumEnabled)
|
|
@@ -117,6 +78,10 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
117
78
|
|
|
118
79
|
references.forEach((slot, i) => {
|
|
119
80
|
const label = slotLabel(slot)
|
|
81
|
+
const persona = slot?.role_persona && ROLE_PERSONA_PROMPTS[slot.role_persona]
|
|
82
|
+
? `\n\n${ROLE_PERSONA_PROMPTS[slot.role_persona]}`
|
|
83
|
+
: ''
|
|
84
|
+
const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
|
|
120
85
|
|
|
121
86
|
const runOne = async () => {
|
|
122
87
|
try {
|
|
@@ -125,7 +90,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
125
90
|
{
|
|
126
91
|
provider: slot.provider,
|
|
127
92
|
model: slot.model,
|
|
128
|
-
messages:
|
|
93
|
+
messages: slotMessages,
|
|
129
94
|
temperature: options.temperature ?? 0.6,
|
|
130
95
|
maxTokens: options.maxTokens ?? 4096,
|
|
131
96
|
timeoutMs,
|
|
@@ -144,7 +109,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
144
109
|
const text = typeof res === 'string' ? res : (res?.content || res?.text || '')
|
|
145
110
|
|
|
146
111
|
const fallbackUsage = (typeof res === 'object' && res?.usage) ? res.usage : {
|
|
147
|
-
inputTokens: Math.max(1, Math.round(
|
|
112
|
+
inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
|
|
148
113
|
outputTokens: Math.max(1, Math.round(text.length / 4)),
|
|
149
114
|
}
|
|
150
115
|
const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
|
|
@@ -327,6 +292,10 @@ export async function runMoAPipeline({
|
|
|
327
292
|
const candidateRetries = 1
|
|
328
293
|
const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
329
294
|
const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
|
|
295
|
+
const isBlindEvaluation = Boolean(preset?.blind_evaluation)
|
|
296
|
+
const isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
|
|
297
|
+
const allowCandidateOverride = Boolean(preset?.allow_candidate_override)
|
|
298
|
+
const runId = crypto.randomUUID()
|
|
330
299
|
|
|
331
300
|
// 2. Collect project context for refinement tasks
|
|
332
301
|
const isRefinement = isRefinementTask(userPrompt)
|
|
@@ -530,8 +499,50 @@ export async function runMoAPipeline({
|
|
|
530
499
|
}
|
|
531
500
|
}
|
|
532
501
|
|
|
502
|
+
// 7b. Consilium Round 2 (Peer Critique) if enabled
|
|
503
|
+
if (isPeerCritiqueEnabled && successfulRefs.length > 1) {
|
|
504
|
+
if (typeof onProgress === 'function') {
|
|
505
|
+
onProgress('🤝 *Consilium Round 2: Candidates reviewing peer proposals in parallel...*\n')
|
|
506
|
+
}
|
|
507
|
+
const r2Promises = successfulRefs.map(async (cand) => {
|
|
508
|
+
const opponents = successfulRefs
|
|
509
|
+
.filter((c) => c.index !== cand.index)
|
|
510
|
+
.map((c) => ({ label: isBlindEvaluation ? `Candidate ${c.index}` : c.label, text: c.text }))
|
|
511
|
+
const r2Prompt = buildPeerCritiquePrompt(userPrompt, cand.text, opponents, isBlindEvaluation)
|
|
512
|
+
try {
|
|
513
|
+
const slot = referenceModels[cand.index - 1] || { provider: cand.provider, model: cand.model }
|
|
514
|
+
const res = await callWithTransientRetry(callLlm, {
|
|
515
|
+
provider: slot.provider,
|
|
516
|
+
model: slot.model,
|
|
517
|
+
messages: [{ role: 'user', content: r2Prompt }],
|
|
518
|
+
temperature: refTemp,
|
|
519
|
+
maxTokens,
|
|
520
|
+
timeoutMs: refTimeoutSec * 1000,
|
|
521
|
+
}, candidateRetries)
|
|
522
|
+
const refinedText = typeof res === 'string' ? res : (res?.content || res?.text || cand.text)
|
|
523
|
+
cand.text = refinedText
|
|
524
|
+
const r2Files = extractFileBlocks(refinedText)
|
|
525
|
+
if (r2Files.length > 0) {
|
|
526
|
+
cand.files = r2Files
|
|
527
|
+
cand.syntaxWarning = (verifyFileSyntax(r2Files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
|
|
528
|
+
}
|
|
529
|
+
if (typeof res === 'object' && res?.usage) {
|
|
530
|
+
cand.usage.inputTokens = (cand.usage.inputTokens || 0) + (res.usage.inputTokens || 0)
|
|
531
|
+
cand.usage.outputTokens = (cand.usage.outputTokens || 0) + (res.usage.outputTokens || 0)
|
|
532
|
+
cand.usage.totalTokens = (cand.usage.totalTokens || 0) + (res.usage.totalTokens || 0)
|
|
533
|
+
const extraCost = estimateTokenCost(slot, res.usage, prices)
|
|
534
|
+
cand.costUsd = Number(((cand.costUsd || 0) + extraCost.costUsd).toFixed(5))
|
|
535
|
+
}
|
|
536
|
+
} catch {}
|
|
537
|
+
})
|
|
538
|
+
await Promise.allSettled(r2Promises)
|
|
539
|
+
}
|
|
540
|
+
|
|
533
541
|
// 8. Synthesis phase via primary judge or fallback chain
|
|
534
|
-
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
542
|
+
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
543
|
+
curatorSynthesis: isCuratorSynthesis,
|
|
544
|
+
blindEvaluation: isBlindEvaluation,
|
|
545
|
+
})
|
|
535
546
|
let synthesizedText = ''
|
|
536
547
|
let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
|
|
537
548
|
let chosenJudge = primaryJudge
|
|
@@ -600,15 +611,16 @@ export async function runMoAPipeline({
|
|
|
600
611
|
const synthesizedFiles = extractFileBlocks(synthesizedText)
|
|
601
612
|
let promotedFiles = []
|
|
602
613
|
|
|
614
|
+
const keepWorkspaces = allowCandidateOverride
|
|
603
615
|
if (synthesizedFiles.length > 0) {
|
|
604
616
|
await writeCandidateWorkspace(cwd, 'curator-synthesis', synthesizedFiles)
|
|
605
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis')
|
|
617
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis', { keepMoa: keepWorkspaces })
|
|
606
618
|
} else if (referenceOutputs[winningIndex - 1]?.files?.length > 0) {
|
|
607
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex)
|
|
619
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex, { keepMoa: keepWorkspaces })
|
|
608
620
|
} else if (successfulRefs[0]?.files?.length > 0) {
|
|
609
621
|
const fallbackIdx = successfulRefs[0].index
|
|
610
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx)
|
|
611
|
-
} else {
|
|
622
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx, { keepMoa: keepWorkspaces })
|
|
623
|
+
} else if (!keepWorkspaces) {
|
|
612
624
|
await cleanMoaWorkspaces(cwd)
|
|
613
625
|
}
|
|
614
626
|
|
|
@@ -625,6 +637,7 @@ export async function runMoAPipeline({
|
|
|
625
637
|
// Record run to history
|
|
626
638
|
try {
|
|
627
639
|
recordMoaRun({
|
|
640
|
+
id: runId,
|
|
628
641
|
prompt: userPrompt,
|
|
629
642
|
preset: preset?.name || 'default',
|
|
630
643
|
isRefinement,
|
|
@@ -654,6 +667,8 @@ export async function runMoAPipeline({
|
|
|
654
667
|
winningIndex,
|
|
655
668
|
winnerModel,
|
|
656
669
|
promotedFiles,
|
|
670
|
+
runId,
|
|
671
|
+
allowCandidateOverride,
|
|
657
672
|
...(livePreview ? { liveCanvas: livePreview } : {}),
|
|
658
673
|
usage: {
|
|
659
674
|
totalTokens,
|