@goodandready/dsh-moa 0.2.25 → 0.2.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +54 -0
- package/lib/history.js +6 -0
- package/lib/index.js +5 -0
- package/lib/moa-prompts.js +22 -6
- package/lib/moa-runner.js +34 -35
- package/lib/moa-test-gate.js +236 -0
- package/package.json +1 -1
package/lib/client.js
CHANGED
|
@@ -63,6 +63,10 @@ window.__ModuleLoader__.load({
|
|
|
63
63
|
'aggregator.peer_hint': 'Candidates critique each other\'s solutions and refine their code before final judge synthesis.',
|
|
64
64
|
'aggregator.override_label': 'Allow Candidate Override Actions',
|
|
65
65
|
'aggregator.override_hint': 'Keeps candidate workspaces in .moa to let you apply any candidate files via chat command.',
|
|
66
|
+
'aggregator.test_gate_label': 'Test Execution Gate (Automatic Sandbox Testing)',
|
|
67
|
+
'aggregator.test_gate_hint': 'Executes test suite inside candidate sandboxes before judge review. Real test results guide winner selection.',
|
|
68
|
+
'aggregator.test_cmd_placeholder': 'e.g. npm test or node --test',
|
|
69
|
+
'aggregator.test_timeout_label': 'Test timeout (sec):',
|
|
66
70
|
'proposers.role_label': 'Persona:',
|
|
67
71
|
'proposers.role_general': 'General',
|
|
68
72
|
'proposers.role_minimalist': 'Minimalist',
|
|
@@ -177,6 +181,10 @@ window.__ModuleLoader__.load({
|
|
|
177
181
|
'aggregator.peer_hint': '在主裁判终审前,各候选模型互相审阅并改进彼此的方案代码。',
|
|
178
182
|
'aggregator.override_label': '允许手动候选方案提拔操作 (Override)',
|
|
179
183
|
'aggregator.override_hint': '在 .moa 目录中保留候选模型的工作区,允许通过聊天命令提拔任意候选模型的文件。',
|
|
184
|
+
'aggregator.test_gate_label': '测试门禁 (沙箱自动化测试)',
|
|
185
|
+
'aggregator.test_gate_hint': '在裁判评审前在候选沙箱中执行测试套件,真实测试结果将指导获胜者评选。',
|
|
186
|
+
'aggregator.test_cmd_placeholder': '例如 npm test 或 node --test',
|
|
187
|
+
'aggregator.test_timeout_label': '测试超时时间 (秒):',
|
|
180
188
|
'proposers.role_label': '工程角色画像:',
|
|
181
189
|
'proposers.role_general': '通用平衡 (General)',
|
|
182
190
|
'proposers.role_minimalist': '极简标准库 (Minimalist)',
|
|
@@ -1273,6 +1281,52 @@ window.__ModuleLoader__.load({
|
|
|
1273
1281
|
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1274
1282
|
t('aggregator.override_hint')
|
|
1275
1283
|
),
|
|
1284
|
+
React.createElement(
|
|
1285
|
+
'label',
|
|
1286
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 8 } },
|
|
1287
|
+
React.createElement('input', {
|
|
1288
|
+
type: 'checkbox',
|
|
1289
|
+
checked: Boolean(currentPreset.test_gate_enabled),
|
|
1290
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, test_gate_enabled: e.target.checked })),
|
|
1291
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1292
|
+
}),
|
|
1293
|
+
t('aggregator.test_gate_label')
|
|
1294
|
+
),
|
|
1295
|
+
React.createElement(
|
|
1296
|
+
'div',
|
|
1297
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1298
|
+
t('aggregator.test_gate_hint')
|
|
1299
|
+
),
|
|
1300
|
+
Boolean(currentPreset.test_gate_enabled) &&
|
|
1301
|
+
React.createElement(
|
|
1302
|
+
'div',
|
|
1303
|
+
{ style: { marginLeft: 22, marginTop: 6, display: 'flex', flexDirection: 'column', gap: 6 } },
|
|
1304
|
+
React.createElement('input', {
|
|
1305
|
+
type: 'text',
|
|
1306
|
+
className: 'moa-input',
|
|
1307
|
+
style: { fontSize: 12, height: 28 },
|
|
1308
|
+
placeholder: t('aggregator.test_cmd_placeholder'),
|
|
1309
|
+
value: currentPreset.test_command || '',
|
|
1310
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, test_command: e.target.value })),
|
|
1311
|
+
}),
|
|
1312
|
+
React.createElement(
|
|
1313
|
+
'div',
|
|
1314
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8 } },
|
|
1315
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('aggregator.test_timeout_label')),
|
|
1316
|
+
React.createElement('input', {
|
|
1317
|
+
type: 'number',
|
|
1318
|
+
className: 'moa-input',
|
|
1319
|
+
style: { width: 70, height: 26, fontSize: 12 },
|
|
1320
|
+
min: 5,
|
|
1321
|
+
max: 120,
|
|
1322
|
+
value: currentPreset.test_gate_timeout_sec ?? 15,
|
|
1323
|
+
onChange: (e) => {
|
|
1324
|
+
const val = parseInt(e.target.value, 10)
|
|
1325
|
+
updateCurrentPreset((p) => ({ ...p, test_gate_timeout_sec: isNaN(val) ? 15 : val }))
|
|
1326
|
+
},
|
|
1327
|
+
})
|
|
1328
|
+
)
|
|
1329
|
+
),
|
|
1276
1330
|
/* Fallback Judges Chain */
|
|
1277
1331
|
React.createElement(
|
|
1278
1332
|
'div',
|
package/lib/history.js
CHANGED
|
@@ -416,5 +416,11 @@ export function candidatesForHistory(referenceOutputs) {
|
|
|
416
416
|
files: (r.files || []).map((f) => f.relativePath),
|
|
417
417
|
usage: r.usage || { inputTokens: 0, outputTokens: 0 },
|
|
418
418
|
costUsd: r.costUsd || 0,
|
|
419
|
+
testResult: r.testResult ? {
|
|
420
|
+
passed: r.testResult.passed,
|
|
421
|
+
exitCode: r.testResult.exitCode,
|
|
422
|
+
summary: r.testResult.summary,
|
|
423
|
+
durationMs: r.testResult.durationMs,
|
|
424
|
+
} : null,
|
|
419
425
|
}))
|
|
420
426
|
}
|
package/lib/index.js
CHANGED
|
@@ -74,6 +74,9 @@ export const PresetSchema = z.object({
|
|
|
74
74
|
allow_candidate_override: z.boolean().default(false),
|
|
75
75
|
max_tokens: z.number().default(4096),
|
|
76
76
|
judge_criteria: z.string().default(''),
|
|
77
|
+
test_gate_enabled: z.boolean().default(false),
|
|
78
|
+
test_command: z.string().default(''),
|
|
79
|
+
test_gate_timeout_sec: z.number().default(15),
|
|
77
80
|
})
|
|
78
81
|
|
|
79
82
|
export const Config = z.object({
|
|
@@ -394,3 +397,5 @@ export function apply(ctx, config) {
|
|
|
394
397
|
})
|
|
395
398
|
}, 'dsh-moa: llm stream interceptor')
|
|
396
399
|
}
|
|
400
|
+
|
|
401
|
+
export { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
package/lib/moa-prompts.js
CHANGED
|
@@ -57,6 +57,12 @@ export const ROLE_PERSONA_PROMPTS = {
|
|
|
57
57
|
general: '',
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
+
export const TEST_GATE_DIRECTIVE = `### 🧪 Test Execution Gate Directive:
|
|
61
|
+
Some or all candidates were evaluated by executing real project tests against their sandbox implementations.
|
|
62
|
+
CRITICAL JUDGE DIRECTIVE: Test execution results are empirical ground truth.
|
|
63
|
+
- A candidate whose code passes all tests (PASS) must be strongly favored over candidates whose code causes test failures or exceptions (FAIL).
|
|
64
|
+
- If a candidate with test failures is chosen due to significantly superior architecture, you MUST explicitly identify and repair the failing test/code in your final synthesis deliverable.`
|
|
65
|
+
|
|
60
66
|
export const SYNTAX_CORRECTION_DIRECTIVE = `### ⚠️ Syntax Warning & Auto-Fix Directive:
|
|
61
67
|
Some candidate proposals have detected syntax flaws (annotated with [⚠️ Syntax Warning]).
|
|
62
68
|
CRITICAL JUDGE DIRECTIVE: If a candidate with a syntax flaw presents superior architecture, algorithm, or engineering logic compared to other candidates, DO NOT reject them solely for this syntax flaw!
|
|
@@ -235,10 +241,14 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
|
|
|
235
241
|
const role = r.role_persona || r.slot?.role_persona || r.role || r.slot?.role
|
|
236
242
|
const roleNote = role && role !== 'general' ? ` [Focus: ${role}]` : ''
|
|
237
243
|
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
244
|
+
const testNote = r.testResult ? ` [🧪 Test Gate: ${r.testResult.summary}]` : ''
|
|
238
245
|
const header = isBlind
|
|
239
|
-
? `Candidate ${i + 1}${roleNote}:${syntaxNote}${fileSummary}`
|
|
240
|
-
: `Candidate ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${fileSummary}`
|
|
241
|
-
|
|
246
|
+
? `Candidate ${i + 1}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
247
|
+
: `Candidate ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
248
|
+
const testReport = r.testResult
|
|
249
|
+
? `\n\n[🧪 Test Execution Gate Report]:\n- Status: ${r.testResult.passed ? 'PASSED ✅' : 'FAILED ❌'} (Exit code: ${r.testResult.exitCode})\n- Duration: ${r.testResult.durationMs}ms\n- Output:\n\`\`\`\n${r.testResult.output}\n\`\`\``
|
|
250
|
+
: ''
|
|
251
|
+
return `${header}\n${textContent}${testReport}`
|
|
242
252
|
})
|
|
243
253
|
.join('\n\n')
|
|
244
254
|
|
|
@@ -258,6 +268,7 @@ ${joined}
|
|
|
258
268
|
${ANTIPATTERNS_RUBRIC}
|
|
259
269
|
|
|
260
270
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
271
|
+
${referenceOutputs.some((r) => r.testResult) ? `\n${TEST_GATE_DIRECTIVE}\n` : ''}
|
|
261
272
|
${referenceOutputs.some((r) => r.syntaxWarning) ? `
|
|
262
273
|
${SYNTAX_CORRECTION_DIRECTIVE}
|
|
263
274
|
` : ''}
|
|
@@ -304,10 +315,14 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
|
|
|
304
315
|
const role = r.role_persona || r.slot?.role_persona || r.role || r.slot?.role
|
|
305
316
|
const roleNote = role && role !== 'general' ? ` [Focus: ${role}]` : ''
|
|
306
317
|
const syntaxNote = r.syntaxWarning ? ` [⚠️ Syntax Warning: ${r.syntaxWarning}]` : ''
|
|
318
|
+
const testNote = r.testResult ? ` [🧪 Test Gate: ${r.testResult.summary}]` : ''
|
|
307
319
|
const header = isBlind
|
|
308
|
-
? `Reference ${i + 1}${roleNote}:${syntaxNote}${fileSummary}`
|
|
309
|
-
: `Reference ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${fileSummary}`
|
|
310
|
-
|
|
320
|
+
? `Reference ${i + 1}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
321
|
+
: `Reference ${i + 1} — ${r.label}${roleNote}:${syntaxNote}${testNote}${fileSummary}`
|
|
322
|
+
const testReport = r.testResult
|
|
323
|
+
? `\n\n[🧪 Test Execution Gate Report]:\n- Status: ${r.testResult.passed ? 'PASSED ✅' : 'FAILED ❌'} (Exit code: ${r.testResult.exitCode})\n- Duration: ${r.testResult.durationMs}ms\n- Output:\n\`\`\`\n${r.testResult.output}\n\`\`\``
|
|
324
|
+
: ''
|
|
325
|
+
return `${header}\n${textContent}${testReport}`
|
|
311
326
|
})
|
|
312
327
|
.join('\n\n')
|
|
313
328
|
|
|
@@ -326,6 +341,7 @@ ${joined}
|
|
|
326
341
|
${ANTIPATTERNS_RUBRIC}
|
|
327
342
|
|
|
328
343
|
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
344
|
+
${referenceOutputs.some((r) => r.testResult) ? `\n${TEST_GATE_DIRECTIVE}\n` : ''}
|
|
329
345
|
${referenceOutputs.some((r) => r.syntaxWarning) ? `\n${SYNTAX_CORRECTION_DIRECTIVE}\n` : ''}
|
|
330
346
|
Instructions:
|
|
331
347
|
Your response MUST be structured into three clear parts:
|
package/lib/moa-runner.js
CHANGED
|
@@ -1,28 +1,21 @@
|
|
|
1
|
-
import { bestEffort } from './best-effort.js'
|
|
2
|
-
/**
|
|
3
|
-
* DeepSeek Harness Mixture of Agents (MoA) — Runner Engine
|
|
4
|
-
* Parallel fan-out, Consilium Round 2 peer critique, aggregator synthesis & streaming.
|
|
5
|
-
*/
|
|
6
|
-
|
|
7
1
|
import path from 'node:path'
|
|
8
2
|
import crypto from 'node:crypto'
|
|
3
|
+
import { bestEffort } from './best-effort.js'
|
|
9
4
|
import { extractFileBlocks, collectProjectContext, formatProjectContext, isRefinementTask, writeCandidateWorkspace, promoteCandidateWorkspace, cleanMoaWorkspaces, verifyFileSyntax } from './file-workspace.js'
|
|
10
5
|
import { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
|
|
11
6
|
import { recordMoaRun, recordMoaRunAsync, candidatesForHistory } from './history.js'
|
|
12
7
|
import { createPromotedPreview } from './live-canvas.js'
|
|
13
8
|
import { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, buildPeerCritiquePrompt, ROLE_PERSONA_PROMPTS, SYSTEM_ROLE_PROPOSER, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
14
9
|
import { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
10
|
+
import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
11
|
+
import { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
|
15
12
|
|
|
16
13
|
// Re-exports for consumers & backward compatibility
|
|
17
14
|
export { slotLabel, cleanAdvisoryMessages, isBroadPromptRequiringQuestions, buildQuestionSynthesisPrompt, buildCuratorSynthesisPrompt, buildSynthesisPrompt, ANTIPATTERNS_RUBRIC } from './moa-prompts.js'
|
|
18
15
|
export { parseWinnerIndex, parseRecommendedAssembler, parseMoACommand, stripOrSummarizeCode, formatMoAResponse } from './moa-parser.js'
|
|
19
16
|
export { estimateTokenCost, summarizeMoAUsage } from './pricing.js'
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
* Invokes LLM call with transient retry for recoverable network/rate-limit errors.
|
|
23
|
-
*/
|
|
24
|
-
import { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
25
|
-
export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel }
|
|
17
|
+
export { REFERENCE_SYSTEM_PROMPT, callWithTransientRetry, runReferencesParallel } from './moa-candidates.js'
|
|
18
|
+
export { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
|
26
19
|
|
|
27
20
|
/**
|
|
28
21
|
* Executes the full Mixture of Agents pipeline.
|
|
@@ -39,18 +32,18 @@ export async function runMoAPipeline({
|
|
|
39
32
|
liveCanvas = null,
|
|
40
33
|
onStreamDelta = null,
|
|
41
34
|
signal = null,
|
|
35
|
+
testGateEnabled = null,
|
|
36
|
+
testCommand = null,
|
|
37
|
+
execFn = null,
|
|
42
38
|
}) {
|
|
43
39
|
if (signal?.aborted) return { error: new Error('Turn aborted') }
|
|
44
40
|
const startTime = Date.now()
|
|
45
41
|
|
|
46
42
|
// 1. Resolve configurations
|
|
47
43
|
const referenceModels = Array.isArray(preset?.reference_models) && preset.reference_models.length > 0
|
|
48
|
-
? preset.reference_models
|
|
49
|
-
: [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
|
|
50
|
-
|
|
44
|
+
? preset.reference_models : [{ provider: 'opencode-go', model: 'deepseek-v4-flash' }]
|
|
51
45
|
const primaryJudge = preset?.aggregator?.provider && preset?.aggregator?.model
|
|
52
|
-
? preset.aggregator
|
|
53
|
-
: { provider: 'codex', model: 'gpt-5.6-sol' }
|
|
46
|
+
? preset.aggregator : { provider: 'codex', model: 'gpt-5.6-sol' }
|
|
54
47
|
|
|
55
48
|
const fallbackJudges = Array.isArray(preset?.aggregator_fallbacks) ? preset.aggregator_fallbacks : []
|
|
56
49
|
const judgesChain = [primaryJudge, ...fallbackJudges]
|
|
@@ -60,17 +53,12 @@ export async function runMoAPipeline({
|
|
|
60
53
|
const maxTokens = typeof preset?.max_tokens === 'number' ? preset.max_tokens : 4096
|
|
61
54
|
const judgeCriteria = preset?.judge_criteria || ''
|
|
62
55
|
const isFastMode = referenceModels.length === 1 && !preset?.curator_synthesis
|
|
63
|
-
const isCuratorSynthesis = Boolean(preset?.curator_synthesis)
|
|
64
|
-
const
|
|
65
|
-
const
|
|
66
|
-
const gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
|
|
67
|
-
const candidateRetries = 1
|
|
68
|
-
const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
56
|
+
const isCuratorSynthesis = Boolean(preset?.curator_synthesis), isStreamAggregator = preset?.stream_aggregator !== false
|
|
57
|
+
const isQuorumEnabled = Boolean(preset?.quorum_enabled), gracePeriodSec = typeof preset?.grace_period_sec === 'number' ? preset.grace_period_sec : 10
|
|
58
|
+
const candidateRetries = 1, refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
69
59
|
const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
|
|
70
|
-
const isBlindEvaluation = Boolean(preset?.blind_evaluation)
|
|
71
|
-
const
|
|
72
|
-
const allowCandidateOverride = Boolean(preset?.allow_candidate_override)
|
|
73
|
-
const runId = crypto.randomUUID()
|
|
60
|
+
const isBlindEvaluation = Boolean(preset?.blind_evaluation), isPeerCritiqueEnabled = Boolean(preset?.peer_critique_enabled)
|
|
61
|
+
const allowCandidateOverride = Boolean(preset?.allow_candidate_override), runId = crypto.randomUUID()
|
|
74
62
|
|
|
75
63
|
// 2. Collect project context for refinement tasks
|
|
76
64
|
const collectedCtx = await collectProjectContext(cwd, 16000)
|
|
@@ -159,10 +147,20 @@ export async function runMoAPipeline({
|
|
|
159
147
|
const files = extractFileBlocks(ref.text)
|
|
160
148
|
ref.files = files
|
|
161
149
|
if (files.length > 0) {
|
|
150
|
+
ref.syntaxWarning = (verifyFileSyntax(files) || []).map((w) => `${w.file}: ${w.error}`).join('; ')
|
|
162
151
|
await writeCandidateWorkspace(cwd, i + 1, files)
|
|
163
152
|
}
|
|
164
153
|
}
|
|
165
154
|
|
|
155
|
+
// 5b. Test Execution Gate
|
|
156
|
+
await executeTestGateForCandidates({
|
|
157
|
+
cwd,
|
|
158
|
+
referenceOutputs,
|
|
159
|
+
preset,
|
|
160
|
+
options: { testGateEnabled, testCommand, execFn },
|
|
161
|
+
onProgress,
|
|
162
|
+
})
|
|
163
|
+
|
|
166
164
|
// 6. Questionnaire synthesis branch
|
|
167
165
|
if (needsQuestions) {
|
|
168
166
|
if (typeof onProgress === 'function') {
|
|
@@ -188,11 +186,7 @@ export async function runMoAPipeline({
|
|
|
188
186
|
qUsage = estimateTokenCost(primaryJudge, qFallbackUsage, prices)
|
|
189
187
|
} catch (err) {
|
|
190
188
|
console.warn('[dsh-moa] Questionnaire synthesis failed, proceeding with fallback questions:', err)
|
|
191
|
-
questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n
|
|
192
|
-
'1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n' +
|
|
193
|
-
'2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n' +
|
|
194
|
-
'3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n' +
|
|
195
|
-
'*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*'
|
|
189
|
+
questionsContent = `### Clarification of Requirements: "${userPrompt}"\n\n1. **Architecture & Scope**: Single-file deliverable or multi-module project structure?\n2. **Design & Style**: Minimalist, dark mode, or clean neutral theme?\n3. **Functional Priorities**: Core MVP or comprehensive extended implementation?\n\n*Reply with your preferences (e.g. "1, 2") or proceed with defaults.*`
|
|
196
190
|
}
|
|
197
191
|
|
|
198
192
|
await cleanMoaWorkspaces(cwd)
|
|
@@ -344,6 +338,13 @@ export async function runMoAPipeline({
|
|
|
344
338
|
}
|
|
345
339
|
})
|
|
346
340
|
await Promise.allSettled(r2Promises)
|
|
341
|
+
await executeTestGateForCandidates({
|
|
342
|
+
cwd,
|
|
343
|
+
referenceOutputs: successfulRefs,
|
|
344
|
+
preset,
|
|
345
|
+
options: { testGateEnabled, testCommand, execFn },
|
|
346
|
+
onProgress,
|
|
347
|
+
})
|
|
347
348
|
}
|
|
348
349
|
|
|
349
350
|
// 8. Synthesis phase via primary judge or fallback chain
|
|
@@ -513,9 +514,7 @@ export async function* streamMoATurn({ targetPreset, userPrompt, messages, callL
|
|
|
513
514
|
callLlm,
|
|
514
515
|
cwd,
|
|
515
516
|
onProgress: pushUpdate,
|
|
516
|
-
onStreamDelta: (delta) =>
|
|
517
|
-
pushUpdate(delta)
|
|
518
|
-
},
|
|
517
|
+
onStreamDelta: (delta) => pushUpdate(delta),
|
|
519
518
|
prices,
|
|
520
519
|
historyFilePath,
|
|
521
520
|
liveCanvas,
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
import os from 'node:os'
|
|
2
|
+
import fs from 'node:fs/promises'
|
|
3
|
+
import fsSync from 'node:fs'
|
|
4
|
+
import path from 'node:path'
|
|
5
|
+
import { spawn } from 'node:child_process'
|
|
6
|
+
import { assertPathContained } from './file-workspace.js'
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Resolves the directory for candidate sandbox.
|
|
10
|
+
*/
|
|
11
|
+
export function resolveCandidateDir(baseDir, candidateIndex) {
|
|
12
|
+
if (!baseDir || candidateIndex === null || candidateIndex === undefined) return null
|
|
13
|
+
const targetStr = String(candidateIndex).trim()
|
|
14
|
+
const sanitizedTarget = targetStr.replace(/[^a-zA-Z0-9_-]/g, '')
|
|
15
|
+
const sub = /^\d+$/.test(sanitizedTarget) ? `candidate-${sanitizedTarget}` : (sanitizedTarget || 'candidate-1')
|
|
16
|
+
const moaRoot = path.join(baseDir, '.moa')
|
|
17
|
+
return assertPathContained(moaRoot, sub)
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Executes a test command in a candidate sandbox or ephemeral test staging directory.
|
|
22
|
+
*
|
|
23
|
+
* @param {object} params
|
|
24
|
+
* @param {string} params.cwd - Project root workspace directory
|
|
25
|
+
* @param {number|string} params.candidateIndex - Index of candidate (1, 2, ...)
|
|
26
|
+
* @param {string} params.testCommand - Shell command to execute (e.g. "npm test", "node --test")
|
|
27
|
+
* @param {number} [params.timeoutMs=15000] - Execution timeout in ms
|
|
28
|
+
* @param {number} [params.maxOutputChars=2000] - Maximum captured output chars
|
|
29
|
+
* @param {function} [params.execFn] - Optional mock/test runner function
|
|
30
|
+
* @returns {Promise<object|null>} Test result object or null if skipped
|
|
31
|
+
*/
|
|
32
|
+
export async function runCandidateTestGate({
|
|
33
|
+
cwd,
|
|
34
|
+
candidateIndex,
|
|
35
|
+
testCommand,
|
|
36
|
+
timeoutMs = 15000,
|
|
37
|
+
maxOutputChars = 2000,
|
|
38
|
+
execFn = null,
|
|
39
|
+
}) {
|
|
40
|
+
if (!cwd || !testCommand || typeof testCommand !== 'string') {
|
|
41
|
+
return null
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const trimmedCmd = testCommand.trim()
|
|
45
|
+
if (!trimmedCmd) return null
|
|
46
|
+
|
|
47
|
+
const candidateDir = resolveCandidateDir(cwd, candidateIndex)
|
|
48
|
+
if (!candidateDir || !fsSync.existsSync(candidateDir)) {
|
|
49
|
+
return null
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const startTime = Date.now()
|
|
53
|
+
let stageDir = null
|
|
54
|
+
let executionDir = candidateDir
|
|
55
|
+
|
|
56
|
+
try {
|
|
57
|
+
const basePkg = path.join(cwd, 'package.json')
|
|
58
|
+
const candPkg = path.join(candidateDir, 'package.json')
|
|
59
|
+
if (fsSync.existsSync(basePkg) && !fsSync.existsSync(candPkg)) {
|
|
60
|
+
stageDir = await fs.mkdtemp(path.join(os.tmpdir(), `moa-test-stage-${candidateIndex}-`))
|
|
61
|
+
|
|
62
|
+
await fs.cp(cwd, stageDir, {
|
|
63
|
+
recursive: true,
|
|
64
|
+
filter: (src) => {
|
|
65
|
+
const rel = path.relative(cwd, src)
|
|
66
|
+
if (!rel) return true
|
|
67
|
+
const parts = rel.split(path.sep)
|
|
68
|
+
if (parts[0] === 'node_modules' || parts[0] === '.git' || parts[0] === '.moa') {
|
|
69
|
+
return false
|
|
70
|
+
}
|
|
71
|
+
return true
|
|
72
|
+
},
|
|
73
|
+
})
|
|
74
|
+
|
|
75
|
+
await fs.cp(candidateDir, stageDir, { recursive: true })
|
|
76
|
+
executionDir = stageDir
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const resolvedCmd = trimmedCmd
|
|
80
|
+
.replaceAll('{candidateDir}', candidateDir)
|
|
81
|
+
.replaceAll('{baseDir}', cwd)
|
|
82
|
+
.replaceAll('{stageDir}', executionDir)
|
|
83
|
+
|
|
84
|
+
let exitCode = 0
|
|
85
|
+
let stdout = ''
|
|
86
|
+
let stderr = ''
|
|
87
|
+
|
|
88
|
+
if (typeof execFn === 'function') {
|
|
89
|
+
const customRes = await execFn({
|
|
90
|
+
cwd: executionDir,
|
|
91
|
+
candidateDir,
|
|
92
|
+
command: resolvedCmd,
|
|
93
|
+
timeoutMs,
|
|
94
|
+
})
|
|
95
|
+
exitCode = customRes?.exitCode ?? (customRes?.passed ? 0 : 1)
|
|
96
|
+
stdout = customRes?.stdout || ''
|
|
97
|
+
stderr = customRes?.stderr || ''
|
|
98
|
+
} else {
|
|
99
|
+
const nodePath = path.join(cwd, 'node_modules')
|
|
100
|
+
const env = {
|
|
101
|
+
...process.env,
|
|
102
|
+
MOA_CANDIDATE_INDEX: String(candidateIndex),
|
|
103
|
+
MOA_CANDIDATE_DIR: candidateDir,
|
|
104
|
+
MOA_BASE_DIR: cwd,
|
|
105
|
+
NODE_PATH: process.env.NODE_PATH ? `${nodePath}:${process.env.NODE_PATH}` : nodePath,
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const runPromise = new Promise((resolve) => {
|
|
109
|
+
let child
|
|
110
|
+
try {
|
|
111
|
+
child = spawn(resolvedCmd, {
|
|
112
|
+
cwd: executionDir,
|
|
113
|
+
shell: true,
|
|
114
|
+
env,
|
|
115
|
+
timeout: timeoutMs,
|
|
116
|
+
})
|
|
117
|
+
} catch (spawnErr) {
|
|
118
|
+
resolve({ exitCode: 1, stdout: '', stderr: spawnErr.message || String(spawnErr) })
|
|
119
|
+
return
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const outChunks = []
|
|
123
|
+
const errChunks = []
|
|
124
|
+
|
|
125
|
+
child.stdout?.on('data', (d) => outChunks.push(d))
|
|
126
|
+
child.stderr?.on('data', (d) => errChunks.push(d))
|
|
127
|
+
|
|
128
|
+
child.on('error', (err) => {
|
|
129
|
+
resolve({ exitCode: 1, stdout: Buffer.concat(outChunks).toString('utf8'), stderr: err.message || String(err) })
|
|
130
|
+
})
|
|
131
|
+
|
|
132
|
+
child.on('close', (code, sig) => {
|
|
133
|
+
const combinedErr = Buffer.concat(errChunks).toString('utf8')
|
|
134
|
+
const finalErr = sig ? `${combinedErr}\nProcess terminated by signal ${sig}`.trim() : combinedErr
|
|
135
|
+
resolve({
|
|
136
|
+
exitCode: code ?? (sig ? 128 : 1),
|
|
137
|
+
stdout: Buffer.concat(outChunks).toString('utf8'),
|
|
138
|
+
stderr: finalErr,
|
|
139
|
+
})
|
|
140
|
+
})
|
|
141
|
+
})
|
|
142
|
+
|
|
143
|
+
const execResult = await runPromise
|
|
144
|
+
exitCode = execResult.exitCode
|
|
145
|
+
stdout = execResult.stdout
|
|
146
|
+
stderr = execResult.stderr
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
const durationMs = Date.now() - startTime
|
|
150
|
+
const combinedOutput = [stdout, stderr].filter(Boolean).join('\n').trim()
|
|
151
|
+
const truncatedOutput = combinedOutput.length > maxOutputChars
|
|
152
|
+
? `${combinedOutput.slice(0, maxOutputChars)}\n...[output truncated by Test Gate]`
|
|
153
|
+
: combinedOutput
|
|
154
|
+
|
|
155
|
+
const passed = exitCode === 0
|
|
156
|
+
const summary = passed
|
|
157
|
+
? `PASS (${durationMs}ms)`
|
|
158
|
+
: `FAIL (code ${exitCode}, ${durationMs}ms)`
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
candidateIndex,
|
|
162
|
+
passed,
|
|
163
|
+
exitCode,
|
|
164
|
+
durationMs,
|
|
165
|
+
command: resolvedCmd,
|
|
166
|
+
output: truncatedOutput,
|
|
167
|
+
summary,
|
|
168
|
+
}
|
|
169
|
+
} catch (err) {
|
|
170
|
+
const durationMs = Date.now() - startTime
|
|
171
|
+
return {
|
|
172
|
+
candidateIndex,
|
|
173
|
+
passed: false,
|
|
174
|
+
exitCode: 1,
|
|
175
|
+
durationMs,
|
|
176
|
+
command: trimmedCmd,
|
|
177
|
+
output: `Test Gate Error: ${err.message || String(err)}`,
|
|
178
|
+
summary: `ERROR (${err.message || 'unknown'})`,
|
|
179
|
+
}
|
|
180
|
+
} finally {
|
|
181
|
+
if (stageDir) {
|
|
182
|
+
try {
|
|
183
|
+
await fs.rm(stageDir, { recursive: true, force: true })
|
|
184
|
+
} catch {}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Runs Test Execution Gate across all successful candidates with generated files.
|
|
191
|
+
*/
|
|
192
|
+
export async function executeTestGateForCandidates({
|
|
193
|
+
cwd,
|
|
194
|
+
referenceOutputs = [],
|
|
195
|
+
preset = {},
|
|
196
|
+
options = {},
|
|
197
|
+
onProgress = null,
|
|
198
|
+
}) {
|
|
199
|
+
const isEnabled = Boolean(options.testGateEnabled ?? preset.test_gate_enabled)
|
|
200
|
+
const testCmd = options.testCommand || preset.test_command
|
|
201
|
+
if (!isEnabled || !testCmd || !cwd) {
|
|
202
|
+
return []
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
const timeoutMs = (preset.test_gate_timeout_sec || 15) * 1000
|
|
206
|
+
const maxOutputChars = preset.test_gate_max_output_chars || 2000
|
|
207
|
+
|
|
208
|
+
if (typeof onProgress === 'function') {
|
|
209
|
+
onProgress('🧪 *Test Execution Gate: Running test suite against candidate sandboxes...*\n')
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
const results = []
|
|
213
|
+
for (const ref of referenceOutputs) {
|
|
214
|
+
if (!ref.ok || !ref.files || ref.files.length === 0) continue
|
|
215
|
+
|
|
216
|
+
const res = await runCandidateTestGate({
|
|
217
|
+
cwd,
|
|
218
|
+
candidateIndex: ref.index,
|
|
219
|
+
testCommand: testCmd,
|
|
220
|
+
timeoutMs,
|
|
221
|
+
maxOutputChars,
|
|
222
|
+
execFn: options.execFn,
|
|
223
|
+
})
|
|
224
|
+
|
|
225
|
+
if (res) {
|
|
226
|
+
ref.testResult = res
|
|
227
|
+
results.push(res)
|
|
228
|
+
if (typeof onProgress === 'function') {
|
|
229
|
+
const icon = res.passed ? '✅' : '❌'
|
|
230
|
+
onProgress(`🧪 *Candidate ${ref.index}: ${icon} ${res.summary}*\n`)
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
return results
|
|
236
|
+
}
|