@goodandready/dsh-moa 0.2.12 → 0.2.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -2
- package/README.ru.md +287 -0
- package/README.zh.md +243 -0
- package/docs/README.ru.md +47 -1
- package/docs/README.zh.md +11 -1
- package/docs/design/DESIGN.md +2 -0
- package/docs/plans/59-power-pack-plan.md +73 -0
- package/lib/client.js +111 -2
- package/lib/file-workspace.js +46 -0
- package/lib/history.js +56 -2
- package/lib/index.js +86 -2
- package/lib/moa-parser.js +24 -0
- package/lib/moa-prompts.js +70 -6
- package/lib/moa-runner.js +63 -52
- package/package.json +1 -1
package/lib/moa-prompts.js
CHANGED
|
@@ -33,6 +33,35 @@ Penalize and strictly downgrade candidates exhibiting any of the following flaws
|
|
|
33
33
|
8. 🚫 Context Amnesia & Regressions:
|
|
34
34
|
- Dropping or breaking previously functioning project features while adding new code.`
|
|
35
35
|
|
|
36
|
+
export const ROLE_PERSONA_PROMPTS = {
|
|
37
|
+
minimalist: `### 🎯 Specialized Engineering Persona: THE MINIMALIST (Ponytail)
|
|
38
|
+
- Prioritize the standard library and native platform capabilities over third-party dependencies.
|
|
39
|
+
- Zero boilerplate, no premature abstractions, interfaces, or unneeded layers.
|
|
40
|
+
- Write the shortest, cleanest working solution with maximum readability.`,
|
|
41
|
+
|
|
42
|
+
robustness: `### 🎯 Specialized Engineering Persona: ROBUSTNESS & DEFENSIVE DESIGN
|
|
43
|
+
- Emphasize paranoid input validation, boundary checking, and comprehensive error handling.
|
|
44
|
+
- Anticipate network failures, null/undefined edge cases, malformed data, and race conditions.
|
|
45
|
+
- Ensure grace under failure and informative, actionable error reporting.`,
|
|
46
|
+
|
|
47
|
+
performance: `### 🎯 Specialized Engineering Persona: HIGH PERFORMANCE & EFFICIENCY
|
|
48
|
+
- Prioritize asymptotic time and space complexity, minimal memory allocations, and zero redundant work.
|
|
49
|
+
- Optimize hot paths, streaming data processing, and efficient algorithmic structures.
|
|
50
|
+
- Document the Big-O complexity and performance characteristics of your solution.`,
|
|
51
|
+
|
|
52
|
+
tester: `### 🎯 Specialized Engineering Persona: TESTABILITY & VERIFICATION
|
|
53
|
+
- Structure code for maximum testability, modular decoupling, and clear side-effect boundaries.
|
|
54
|
+
- Include comprehensive unit/integration test specifications verifying happy paths and edge cases.
|
|
55
|
+
- Emphasize deterministic behavior and assertions.`,
|
|
56
|
+
|
|
57
|
+
general: '',
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export const SYNTAX_CORRECTION_DIRECTIVE = `### ⚠️ Syntax Warning & Auto-Fix Directive:
|
|
61
|
+
Some candidate proposals have detected syntax flaws (annotated with [⚠️ Syntax Warning]).
|
|
62
|
+
CRITICAL JUDGE DIRECTIVE: If a candidate with a syntax flaw presents superior architecture, algorithm, or engineering logic compared to other candidates, DO NOT reject them solely for this syntax flaw!
|
|
63
|
+
Instead, you MUST correct the syntax error directly during synthesis/assembly and declare that candidate the winner.`
|
|
64
|
+
|
|
36
65
|
export const LANGUAGE_MIRRORING_DIRECTIVE = `### 🌐 Language Mirroring Requirement
|
|
37
66
|
CRITICAL: You MUST write your entire analysis, reasoning, verdict, explanations and instructions in the EXACT SAME LANGUAGE as the user's prompt (e.g. Russian if the prompt is written in Russian, English if in English). Do NOT switch or translate to English unless explicitly requested by the user.`
|
|
38
67
|
|
|
@@ -154,6 +183,37 @@ At the end, add a note that the user can answer briefly (e.g.: "1, 2, dark theme
|
|
|
154
183
|
/**
|
|
155
184
|
* Builds the curator synthesis prompt with component analysis, model recommendation, and optional blind review.
|
|
156
185
|
*/
|
|
186
|
+
/**
|
|
187
|
+
* Builds the Round 2 Consilium / Peer Critique prompt for candidate models.
|
|
188
|
+
*/
|
|
189
|
+
export function buildPeerCritiquePrompt(userPrompt, myProposal, otherProposals = [], isBlind = true) {
|
|
190
|
+
const othersText = otherProposals
|
|
191
|
+
.map((p, idx) => {
|
|
192
|
+
const label = isBlind ? `Candidate ${idx + 1}` : (p.label || `Candidate ${idx + 1}`)
|
|
193
|
+
return `### ${label} Alternative Proposal:
|
|
194
|
+
${p.text}`
|
|
195
|
+
})
|
|
196
|
+
.join('\n\n---\n\n')
|
|
197
|
+
|
|
198
|
+
return `You are participating in Round 2 (Consilium / Peer Critique & Solution Refinement) of an MoA ensemble.
|
|
199
|
+
|
|
200
|
+
Original User Request:
|
|
201
|
+
${userPrompt}
|
|
202
|
+
|
|
203
|
+
Your Initial Solution (Round 1):
|
|
204
|
+
${myProposal}
|
|
205
|
+
|
|
206
|
+
Alternative Solutions from Peer Candidates:
|
|
207
|
+
${othersText}
|
|
208
|
+
|
|
209
|
+
Instructions:
|
|
210
|
+
1. 🔍 Peer Critique: Briefly evaluate the other candidates' solutions. Point out any hidden bugs, race conditions, antipatterns, or edge-case omissions in their logic.
|
|
211
|
+
2. 🛡️ Defend & Refine: Adopt the strongest architectural ideas or optimizations from competitors to enhance your own solution. Fix any oversights in your own code.
|
|
212
|
+
3. 🚀 Final Refined Deliverable: Present your complete, production-ready, improved code with full explanations.
|
|
213
|
+
|
|
214
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}`
|
|
215
|
+
}
|
|
216
|
+
|
|
157
217
|
export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
|
|
158
218
|
const isBlind = Boolean(options.blindEvaluation)
|
|
159
219
|
|
|
@@ -166,9 +226,10 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
|
|
|
166
226
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
167
227
|
textContent = stripOrSummarizeCode(textContent)
|
|
168
228
|
}
|
|
229
|
+
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
169
230
|
const header = isBlind
|
|
170
|
-
? `Candidate ${i + 1}:${fileSummary}`
|
|
171
|
-
: `Candidate ${i + 1} — ${r.label}:${fileSummary}`
|
|
231
|
+
? `Candidate ${i + 1}:${syntaxNote}${fileSummary}`
|
|
232
|
+
: `Candidate ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
|
|
172
233
|
return `${header}\n${textContent}`
|
|
173
234
|
})
|
|
174
235
|
.join('\n\n')
|
|
@@ -189,7 +250,9 @@ ${joined}
|
|
|
189
250
|
${ANTIPATTERNS_RUBRIC}
|
|
190
251
|
|
|
191
252
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
192
|
-
|
|
253
|
+
${referenceOutputs.some((r) => r.syntaxWarning) ? `
|
|
254
|
+
${SYNTAX_CORRECTION_DIRECTIVE}
|
|
255
|
+
` : ''}
|
|
193
256
|
Instructions:
|
|
194
257
|
Your response MUST be structured into three clear parts:
|
|
195
258
|
|
|
@@ -230,9 +293,10 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
|
|
|
230
293
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
231
294
|
textContent = stripOrSummarizeCode(textContent)
|
|
232
295
|
}
|
|
296
|
+
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
233
297
|
const header = isBlind
|
|
234
|
-
? `Reference ${i + 1}:${fileSummary}`
|
|
235
|
-
: `Reference ${i + 1} — ${r.label}:${fileSummary}`
|
|
298
|
+
? `Reference ${i + 1}:${syntaxNote}${fileSummary}`
|
|
299
|
+
: `Reference ${i + 1} — ${r.label}:${syntaxNote}${fileSummary}`
|
|
236
300
|
return `${header}\n${textContent}`
|
|
237
301
|
})
|
|
238
302
|
.join('\n\n')
|
|
@@ -252,7 +316,7 @@ ${joined}
|
|
|
252
316
|
${ANTIPATTERNS_RUBRIC}
|
|
253
317
|
|
|
254
318
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
255
|
-
|
|
319
|
+
${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
|
|
256
320
|
Instructions:
|
|
257
321
|
Your response MUST be structured into three clear parts:
|
|
258
322
|
|
package/lib/moa-runner.js
CHANGED
|
@@ -11,53 +11,16 @@
|
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
13
|
import path from 'node:path'
|
|
14
|
-
import
|
|
15
|
-
|
|
16
|
-
collectProjectContext,
|
|
17
|
-
isRefinementTask,
|
|
18
|
-
writeCandidateWorkspace,
|
|
19
|
-
promoteCandidateWorkspace,
|
|
20
|
-
cleanMoaWorkspaces,
|
|
21
|
-
} from './file-workspace.js'
|
|
14
|
+
import crypto from 'node:crypto'
|
|
15
|
+
import { extractFileBlocks, collectProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
|
|
22
16
|
import { estimateTokenCost } from './pricing.js'
|
|
23
17
|
import { recordMoaRun, recordMoaRunAsync } from './history.js'
|
|
24
|
-
import {
|
|
25
|
-
|
|
26
|
-
cleanAdvisoryMessages,
|
|
27
|
-
isBroadPromptRequiringQuestions,
|
|
28
|
-
buildQuestionSynthesisPrompt,
|
|
29
|
-
buildCuratorSynthesisPrompt,
|
|
30
|
-
buildSynthesisPrompt,
|
|
31
|
-
SYSTEM_ROLE_PROPOSER,
|
|
32
|
-
ANTIPATTERNS_RUBRIC,
|
|
33
|
-
} from './moa-prompts.js'
|
|
34
|
-
import {
|
|
35
|
-
parseWinnerIndex,
|
|
36
|
-
parseRecommendedAssembler,
|
|
37
|
-
parseMoACommand,
|
|
38
|
-
stripOrSummarizeCode,
|
|
39
|
-
formatMoAResponse,
|
|
40
|
-
} from './moa-parser.js'
|
|
41
|
-
|
|
42
|
-
// Re-export prompt and parser functions for external consumers / backwards compatibility
|
|
43
|
-
export {
|
|
44
|
-
slotLabel,
|
|
45
|
-
cleanAdvisoryMessages,
|
|
46
|
-
isBroadPromptRequiringQuestions,
|
|
47
|
-
buildQuestionSynthesisPrompt,
|
|
48
|
-
buildCuratorSynthesisPrompt,
|
|
49
|
-
buildSynthesisPrompt,
|
|
50
|
-
ANTIPATTERNS_RUBRIC,
|
|
51
|
-
} from './moa-prompts.js'
|
|
52
|
-
|
|
53
|
-
export {
|
|
54
|
-
parseWinnerIndex,
|
|
55
|
-
parseRecommendedAssembler,
|
|
56
|
-
parseMoACommand,
|
|
57
|
-
stripOrSummarizeCode,
|
|
58
|
-
formatMoAResponse,
|
|
59
|
-
} from './moa-parser.js'
|
|
18
|
+
import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
19
|
+
import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
60
20
|
|
|
21
|
+
// Re-exports for consumers & backward compatibility
|
|
22
|
+
export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
23
|
+
export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
61
24
|
export { estimateTokenCost } from './pricing.js'
|
|
62
25
|
|
|
63
26
|
export const DEFAULT_PROMPT = 'Please propose an optimal, well-structured, production-ready solution with full code and explanations.'
|
|
@@ -94,8 +57,6 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
94
57
|
|
|
95
58
|
const advisoryMessages = cleanAdvisoryMessages(messages)
|
|
96
59
|
const systemPrompt = options.systemPrompt || REFERENCE_SYSTEM_PROMPT
|
|
97
|
-
// Deterministic prefix for optimal prompt caching hit rate
|
|
98
|
-
const fullMessages = [{ role: 'system', content: systemPrompt }, ...advisoryMessages]
|
|
99
60
|
const timeoutMs = options.timeoutMs ?? ((options.timeoutSec ?? 60) * 1000)
|
|
100
61
|
const maxRetries = options.maxRetries ?? 0
|
|
101
62
|
const quorumEnabled = Boolean(options.quorumEnabled)
|
|
@@ -117,6 +78,10 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
117
78
|
|
|
118
79
|
references.forEach((slot, i) => {
|
|
119
80
|
const label = slotLabel(slot)
|
|
81
|
+
const persona = slot?.role_persona && ROLE_PERSONA_PROMPTS[slot.role_persona]
|
|
82
|
+
? `\n\n${ROLE_PERSONA_PROMPTS[slot.role_persona]}`
|
|
83
|
+
: ''
|
|
84
|
+
const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
|
|
120
85
|
|
|
121
86
|
const runOne = async () => {
|
|
122
87
|
try {
|
|
@@ -125,7 +90,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
125
90
|
{
|
|
126
91
|
provider: slot.provider,
|
|
127
92
|
model: slot.model,
|
|
128
|
-
messages:
|
|
93
|
+
messages: slotMessages,
|
|
129
94
|
temperature: options.temperature ?? 0.6,
|
|
130
95
|
maxTokens: options.maxTokens ?? 4096,
|
|
131
96
|
timeoutMs,
|
|
@@ -144,7 +109,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
144
109
|
const text = typeof res === 'string' ? res : (res?.content || res?.text || '')
|
|
145
110
|
|
|
146
111
|
const fallbackUsage = (typeof res === 'object' && res?.usage) ? res.usage : {
|
|
147
|
-
inputTokens: Math.max(1, Math.round(
|
|
112
|
+
inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
|
|
148
113
|
outputTokens: Math.max(1, Math.round(text.length / 4)),
|
|
149
114
|
}
|
|
150
115
|
const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
|
|
@@ -328,6 +293,9 @@ export async function runMoAPipeline({
|
|
|
328
293
|
const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
329
294
|
const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
|
|
330
295
|
const isBlindEvaluation = Boolean(preset?.blind_evaluation)
|
|
296
|
+
const isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
|
|
297
|
+
const allowCandidateOverride = Boolean(preset?.allow_candidate_override)
|
|
298
|
+
const runId = crypto.randomUUID()
|
|
331
299
|
|
|
332
300
|
// 2. Collect project context for refinement tasks
|
|
333
301
|
const isRefinement = isRefinementTask(userPrompt)
|
|
@@ -531,6 +499,45 @@ export async function runMoAPipeline({
|
|
|
531
499
|
}
|
|
532
500
|
}
|
|
533
501
|
|
|
502
|
+
// 7b. Consilium Round 2 (Peer Critique) if enabled
|
|
503
|
+
if (isPeerCritiqueEnabled && successfulRefs.length > 1) {
|
|
504
|
+
if (typeof onProgress === 'function') {
|
|
505
|
+
onProgress('🤝 *Consilium Round 2: Candidates reviewing peer proposals in parallel...*\n')
|
|
506
|
+
}
|
|
507
|
+
const r2Promises = successfulRefs.map(async (cand) => {
|
|
508
|
+
const opponents = successfulRefs
|
|
509
|
+
.filter((c) => c.index !== cand.index)
|
|
510
|
+
.map((c) => ({ label: isBlindEvaluation ? `Candidate ${c.index}` : c.label, text: c.text }))
|
|
511
|
+
const r2Prompt = buildPeerCritiquePrompt(userPrompt, cand.text, opponents, isBlindEvaluation)
|
|
512
|
+
try {
|
|
513
|
+
const slot = referenceModels[cand.index - 1] || { provider: cand.provider, model: cand.model }
|
|
514
|
+
const res = await callWithTransientRetry(callLlm, {
|
|
515
|
+
provider: slot.provider,
|
|
516
|
+
model: slot.model,
|
|
517
|
+
messages: [{ role: 'user', content: r2Prompt }],
|
|
518
|
+
temperature: refTemp,
|
|
519
|
+
maxTokens,
|
|
520
|
+
timeoutMs: refTimeoutSec * 1000,
|
|
521
|
+
}, candidateRetries)
|
|
522
|
+
const refinedText = typeof res === 'string' ? res : (res?.content || res?.text || cand.text)
|
|
523
|
+
cand.text = refinedText
|
|
524
|
+
const r2Files = extractFileBlocks(refinedText)
|
|
525
|
+
if (r2Files.length > 0) {
|
|
526
|
+
cand.files = r2Files
|
|
527
|
+
cand.syntaxWarning = (verifyFileSyntax(r2Files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
|
|
528
|
+
}
|
|
529
|
+
if (typeof res === 'object' && res?.usage) {
|
|
530
|
+
cand.usage.inputTokens = (cand.usage.inputTokens || 0) + (res.usage.inputTokens || 0)
|
|
531
|
+
cand.usage.outputTokens = (cand.usage.outputTokens || 0) + (res.usage.outputTokens || 0)
|
|
532
|
+
cand.usage.totalTokens = (cand.usage.totalTokens || 0) + (res.usage.totalTokens || 0)
|
|
533
|
+
const extraCost = estimateTokenCost(slot, res.usage, prices)
|
|
534
|
+
cand.costUsd = Number(((cand.costUsd || 0) + extraCost.costUsd).toFixed(5))
|
|
535
|
+
}
|
|
536
|
+
} catch {}
|
|
537
|
+
})
|
|
538
|
+
await Promise.allSettled(r2Promises)
|
|
539
|
+
}
|
|
540
|
+
|
|
534
541
|
// 8. Synthesis phase via primary judge or fallback chain
|
|
535
542
|
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
536
543
|
curatorSynthesis: isCuratorSynthesis,
|
|
@@ -604,15 +611,16 @@ export async function runMoAPipeline({
|
|
|
604
611
|
const synthesizedFiles = extractFileBlocks(synthesizedText)
|
|
605
612
|
let promotedFiles = []
|
|
606
613
|
|
|
614
|
+
const keepWorkspaces = allowCandidateOverride
|
|
607
615
|
if (synthesizedFiles.length > 0) {
|
|
608
616
|
await writeCandidateWorkspace(cwd, 'curator-synthesis', synthesizedFiles)
|
|
609
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis')
|
|
617
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, 'curator-synthesis', { keepMoa: keepWorkspaces })
|
|
610
618
|
} else if (referenceOutputs[winningIndex - 1]?.files?.length > 0) {
|
|
611
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex)
|
|
619
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, winningIndex, { keepMoa: keepWorkspaces })
|
|
612
620
|
} else if (successfulRefs[0]?.files?.length > 0) {
|
|
613
621
|
const fallbackIdx = successfulRefs[0].index
|
|
614
|
-
promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx)
|
|
615
|
-
} else {
|
|
622
|
+
promotedFiles = await promoteCandidateWorkspace(cwd, fallbackIdx, { keepMoa: keepWorkspaces })
|
|
623
|
+
} else if (!keepWorkspaces) {
|
|
616
624
|
await cleanMoaWorkspaces(cwd)
|
|
617
625
|
}
|
|
618
626
|
|
|
@@ -629,6 +637,7 @@ export async function runMoAPipeline({
|
|
|
629
637
|
// Record run to history
|
|
630
638
|
try {
|
|
631
639
|
recordMoaRun({
|
|
640
|
+
id: runId,
|
|
632
641
|
prompt: userPrompt,
|
|
633
642
|
preset: preset?.name || 'default',
|
|
634
643
|
isRefinement,
|
|
@@ -658,6 +667,8 @@ export async function runMoAPipeline({
|
|
|
658
667
|
winningIndex,
|
|
659
668
|
winnerModel,
|
|
660
669
|
promotedFiles,
|
|
670
|
+
runId,
|
|
671
|
+
allowCandidateOverride,
|
|
661
672
|
...(livePreview ? { liveCanvas: livePreview } : {}),
|
|
662
673
|
usage: {
|
|
663
674
|
totalTokens,
|