@goodandready/dsh-moa 0.2.11 → 0.2.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/docs/README.ru.md +1 -0
- package/docs/design/DESIGN.md +1 -0
- package/lib/client.js +116 -0
- package/lib/index.js +1 -0
- package/lib/moa-prompts.js +29 -10
- package/lib/moa-runner.js +5 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -199,6 +199,7 @@ dsh-moa:
|
|
|
199
199
|
| `presets[].quorum_enabled` | `boolean` | `false` | Straggler mitigation: proceed with synthesis once >= 60% candidates respond |
|
|
200
200
|
| `presets[].grace_period_sec` | `number` | `10` | Grace period in seconds to wait for stragglers after quorum is reached |
|
|
201
201
|
| `presets[].aggregator_fallbacks` | `array` | `[]` | Ordered fallback judge models tried if primary aggregator encounters transient errors |
|
|
202
|
+
| `presets[].blind_evaluation` | `boolean` | `false` | Anonymize candidate model names for the judge/curator to eliminate family/brand bias |
|
|
202
203
|
| `presets[].reference_timeout_sec` | `number` | `60` | Per-candidate execution timeout in seconds |
|
|
203
204
|
| `presets[].aggregator_timeout_sec` | `number` | `180` | Aggregator/judge synthesis timeout in seconds |
|
|
204
205
|
| `presets[].reference_temperature` / `.aggregator_temperature` | `number` | `0.6` / `0.4` | Sampling temperatures for proposers and judge |
|
package/docs/README.ru.md
CHANGED
|
@@ -199,6 +199,7 @@ dsh-moa:
|
|
|
199
199
|
| `presets[].quorum_enabled` | `boolean` | `false` | Защита от зависших моделей (stragglers): запуск синтеза при ответе от >= 60% кандидатов |
|
|
200
200
|
| `presets[].grace_period_sec` | `number` | `10` | Грейс-период (в секундах) ожидания оставшихся моделей после достижения кворума |
|
|
201
201
|
| `presets[].aggregator_fallbacks` | `array` | `[]` | Список запасных моделей-судей при сбоях основной модели агрегатора |
|
|
202
|
+
| `presets[].blind_evaluation` | `boolean` | `false` | Обезличивание имен кандидатов («Candidate 1», «Candidate 2») для исключения предвзятости судьи |
|
|
202
203
|
| `presets[].reference_timeout_sec` | `number` | `60` | Таймаут опроса каждого кандидата в секундах |
|
|
203
204
|
| `presets[].aggregator_timeout_sec` | `number` | `180` | Таймаут синтеза решения судьей в секундах |
|
|
204
205
|
| `presets[].reference_temperature` / `.aggregator_temperature` | `number` | `0.6` / `0.4` | Температуры сэмплирования советников и судьи |
|
package/docs/design/DESIGN.md
CHANGED
|
@@ -58,6 +58,7 @@
|
|
|
58
58
|
- 2026-09-10 — Унификация UI с дизайн-стандартом `dsh-clinebot` (статусные чипы в header, секционные карточки, дизайн-токены `--dsw-alias-*`, глубокий аудит устойчивости).
|
|
59
59
|
- 2026-09-11 — English-canonical пользовательские строки (сервер и клиент); ru-перевод предоставляет translation-плагин, собственный ru-дубль из пакета удалён. Причина: стандарт DSH (dsh-plugin-authoring); пересмотр — только по явному решению владельца.
|
|
60
60
|
- 2026-09-11 — Заглушка Live Canvas удалена (фабрикация `http://localhost:3000/preview/...`); заявления README сняты. Реальная интеграция с `dsh-live-canvas` — отдельная задача (Gitea #46). Changed: прежний пункт про автопревью больше не действует.
|
|
61
|
+
- 2026-09-12 — Реализован режим слепого судейства (`blind_evaluation: boolean`), гарантированное зеркалирование языка запроса (Language Mirroring Guard в промптах судьи и куратора), интерактивный лидерборд моделей (винрейт, запуски, победы, средняя стоимость) в секции аналитики UI и раздельные настройки таймаутов кандидатов и судьи в карточке настроек (Gitea Issue #57).
|
|
61
62
|
- 2026-09-11 — Статусный бейдж карточки отражает фактический `GET /dsh-moa/status` (online / offline / disabled); телеметрия Total Runs / Avg Run Cost берётся из `GET /dsh-moa/history`. Причина: карточка не должна показывать состояния, которые она не проверяла.
|
|
62
63
|
- 2026-09-11 — Переключатель `enabled` в карточке сохраняется через `settingsScope`/REST и влияет на `/moa`-turn и `POST /dsh-moa/run`.
|
|
63
64
|
- 2026-09-11 (вечер) — Интеграция с Live Canvas реализована через собственный REST-контракт `@goodandready/dsh-live-canvas` (`POST /dsh-live-canvas/api/preview`, self-call на порт хоста `ctx.webServer.port`) с тихой деградацией при отсутствии плагина (любая ошибка → ответ без preview-ссылки). Заменяет прежнее решение об удалении фабрикованной заглушки: ссылка `/dsh-live-canvas/sandbox/<id>` теперь создаётся реальной песочницей.
|
package/lib/client.js
CHANGED
|
@@ -55,6 +55,18 @@ window.__ModuleLoader__.load({
|
|
|
55
55
|
'aggregator.curator_hint': 'When enabled: the curator highlights the finest components of each candidate, applies the antipatterns rubric, and advises which agent model should assemble the solution.',
|
|
56
56
|
'aggregator.stream_label': 'Live Stream Curator/Judge Thinking',
|
|
57
57
|
'aggregator.stream_hint': 'Streams aggregator tokens directly to chat in real time for zero-latency initial response.',
|
|
58
|
+
'aggregator.blind_label': 'Blind Review (Anonymize Candidates for Judge)',
|
|
59
|
+
'aggregator.blind_hint': 'When enabled: candidate model names and providers are hidden from the judge prompt to prevent family/brand bias.',
|
|
60
|
+
'aggregator.timeout_label': 'Judge synthesis timeout (sec):',
|
|
61
|
+
'proposers.timeout_label': 'Candidate timeout (sec):',
|
|
62
|
+
'leaderboard.title': '🏆 Model Win-Rate Leaderboard',
|
|
63
|
+
'leaderboard.desc': 'Performance metrics calculated from recorded MoA run verdicts.',
|
|
64
|
+
'leaderboard.model': 'Model',
|
|
65
|
+
'leaderboard.runs': 'Runs',
|
|
66
|
+
'leaderboard.wins': 'Wins',
|
|
67
|
+
'leaderboard.winrate': 'Win Rate',
|
|
68
|
+
'leaderboard.cost': 'Avg Cost',
|
|
69
|
+
'leaderboard.empty': 'No historical runs recorded yet.',
|
|
58
70
|
'aggregator.fallbacks_title': 'Fallback Judges Chain',
|
|
59
71
|
'aggregator.fallbacks_desc': 'Sequential backup models invoked automatically if the primary judge experiences rate limits or outages.',
|
|
60
72
|
'aggregator.add_fallback_btn': '+ Add Fallback Judge',
|
|
@@ -448,6 +460,7 @@ window.__ModuleLoader__.load({
|
|
|
448
460
|
setEnabled,
|
|
449
461
|
hostStatus,
|
|
450
462
|
stats,
|
|
463
|
+
leaderboard = [],
|
|
451
464
|
handleSave,
|
|
452
465
|
saveStatus,
|
|
453
466
|
reload,
|
|
@@ -831,6 +844,39 @@ window.__ModuleLoader__.load({
|
|
|
831
844
|
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
832
845
|
t('aggregator.stream_hint')
|
|
833
846
|
),
|
|
847
|
+
React.createElement(
|
|
848
|
+
'label',
|
|
849
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 8 } },
|
|
850
|
+
React.createElement('input', {
|
|
851
|
+
type: 'checkbox',
|
|
852
|
+
checked: Boolean(currentPreset.blind_evaluation),
|
|
853
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, blind_evaluation: e.target.checked })),
|
|
854
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
855
|
+
}),
|
|
856
|
+
t('aggregator.blind_label')
|
|
857
|
+
),
|
|
858
|
+
React.createElement(
|
|
859
|
+
'div',
|
|
860
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
861
|
+
t('aggregator.blind_hint')
|
|
862
|
+
),
|
|
863
|
+
React.createElement(
|
|
864
|
+
'div',
|
|
865
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, marginTop: 8 } },
|
|
866
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('aggregator.timeout_label')),
|
|
867
|
+
React.createElement('input', {
|
|
868
|
+
type: 'number',
|
|
869
|
+
className: 'moa-input',
|
|
870
|
+
style: { width: 80, height: 28, padding: '0 6px', fontSize: 12 },
|
|
871
|
+
min: 30,
|
|
872
|
+
max: 600,
|
|
873
|
+
value: currentPreset.aggregator_timeout_sec ?? 180,
|
|
874
|
+
onChange: (e) => {
|
|
875
|
+
const val = parseInt(e.target.value, 10)
|
|
876
|
+
updateCurrentPreset((p) => ({ ...p, aggregator_timeout_sec: isNaN(val) ? 180 : val }))
|
|
877
|
+
},
|
|
878
|
+
})
|
|
879
|
+
),
|
|
834
880
|
/* Fallback Judges Chain */
|
|
835
881
|
React.createElement(
|
|
836
882
|
'div',
|
|
@@ -968,6 +1014,23 @@ window.__ModuleLoader__.load({
|
|
|
968
1014
|
updateCurrentPreset((p) => ({ ...p, grace_period_sec: isNaN(val) ? 10 : val }))
|
|
969
1015
|
},
|
|
970
1016
|
})
|
|
1017
|
+
),
|
|
1018
|
+
React.createElement(
|
|
1019
|
+
'div',
|
|
1020
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, marginTop: 4 } },
|
|
1021
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('proposers.timeout_label')),
|
|
1022
|
+
React.createElement('input', {
|
|
1023
|
+
type: 'number',
|
|
1024
|
+
className: 'moa-input',
|
|
1025
|
+
style: { width: 70, height: 28, padding: '0 6px', fontSize: 12 },
|
|
1026
|
+
min: 10,
|
|
1027
|
+
max: 300,
|
|
1028
|
+
value: currentPreset.reference_timeout_sec ?? 60,
|
|
1029
|
+
onChange: (e) => {
|
|
1030
|
+
const val = parseInt(e.target.value, 10)
|
|
1031
|
+
updateCurrentPreset((p) => ({ ...p, reference_timeout_sec: isNaN(val) ? 60 : val }))
|
|
1032
|
+
},
|
|
1033
|
+
})
|
|
971
1034
|
)
|
|
972
1035
|
),
|
|
973
1036
|
React.createElement(
|
|
@@ -1072,6 +1135,48 @@ window.__ModuleLoader__.load({
|
|
|
1072
1135
|
React.createElement('div', { className: 'moa-stat-val' }, String(currentRefs.length)),
|
|
1073
1136
|
React.createElement('div', { className: 'moa-stat-lbl' }, t('stats.active_models'))
|
|
1074
1137
|
)
|
|
1138
|
+
),
|
|
1139
|
+
Array.isArray(leaderboard) && leaderboard.length > 0 && React.createElement(
|
|
1140
|
+
'div',
|
|
1141
|
+
{ style: { marginTop: 12, display: 'flex', flexDirection: 'column', gap: 6 } },
|
|
1142
|
+
React.createElement('div', { style: { fontSize: 13, fontWeight: 600, color: 'var(--dsw-alias-label-primary)' } }, t('leaderboard.title')),
|
|
1143
|
+
React.createElement('div', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginTop: -4 } }, t('leaderboard.desc')),
|
|
1144
|
+
React.createElement(
|
|
1145
|
+
'div',
|
|
1146
|
+
{ style: { overflowX: 'auto', border: '1px solid var(--dsw-alias-border-l2)', borderRadius: 8 } },
|
|
1147
|
+
React.createElement(
|
|
1148
|
+
'table',
|
|
1149
|
+
{ style: { width: '100%', borderCollapse: 'collapse', fontSize: 12, textAlign: 'left' } },
|
|
1150
|
+
React.createElement(
|
|
1151
|
+
'thead',
|
|
1152
|
+
{ style: { background: 'var(--dsw-alias-bg-layer-2)', borderBottom: '1px solid var(--dsw-alias-border-l2)' } },
|
|
1153
|
+
React.createElement(
|
|
1154
|
+
'tr',
|
|
1155
|
+
null,
|
|
1156
|
+
React.createElement('th', { style: { padding: '6px 10px' } }, t('leaderboard.model')),
|
|
1157
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.runs')),
|
|
1158
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.wins')),
|
|
1159
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.winrate')),
|
|
1160
|
+
React.createElement('th', { style: { padding: '6px 10px' } }, t('leaderboard.cost'))
|
|
1161
|
+
)
|
|
1162
|
+
),
|
|
1163
|
+
React.createElement(
|
|
1164
|
+
'tbody',
|
|
1165
|
+
null,
|
|
1166
|
+
leaderboard.map((m, mIdx) =>
|
|
1167
|
+
React.createElement(
|
|
1168
|
+
'tr',
|
|
1169
|
+
{ key: mIdx, style: { borderBottom: mIdx < leaderboard.length - 1 ? '1px solid var(--dsw-alias-border-l2)' : 'none' } },
|
|
1170
|
+
React.createElement('td', { style: { padding: '6px 10px', fontWeight: 500, color: 'var(--dsw-alias-label-primary)' } }, m.modelKey || m.model),
|
|
1171
|
+
React.createElement('td', { style: { padding: '6px 8px', color: 'var(--dsw-alias-label-secondary)' } }, String(m.runs)),
|
|
1172
|
+
React.createElement('td', { style: { padding: '6px 8px', color: 'var(--dsw-alias-state-success-primary)' } }, String(m.wins)),
|
|
1173
|
+
React.createElement('td', { style: { padding: '6px 8px', fontWeight: 600 } }, (m.winRate ?? 0) + '%'),
|
|
1174
|
+
React.createElement('td', { style: { padding: '6px 10px', color: 'var(--dsw-alias-label-secondary)' } }, (m.avgCostUsd > 0) ? ('$' + m.avgCostUsd.toFixed(4)) : 'Free')
|
|
1175
|
+
)
|
|
1176
|
+
)
|
|
1177
|
+
)
|
|
1178
|
+
)
|
|
1179
|
+
)
|
|
1075
1180
|
)
|
|
1076
1181
|
),
|
|
1077
1182
|
|
|
@@ -1151,6 +1256,7 @@ window.__ModuleLoader__.load({
|
|
|
1151
1256
|
const [enabled, setEnabled] = React.useState(true)
|
|
1152
1257
|
const [hostStatus, setHostStatus] = React.useState('connecting')
|
|
1153
1258
|
const [stats, setStats] = React.useState({ totalRuns: null, avgCostUsd: null })
|
|
1259
|
+
const [leaderboard, setLeaderboard] = React.useState([])
|
|
1154
1260
|
|
|
1155
1261
|
React.useEffect(() => {
|
|
1156
1262
|
if (!scope || !snapshot) return
|
|
@@ -1182,6 +1288,15 @@ window.__ModuleLoader__.load({
|
|
|
1182
1288
|
setHostStatus('offline')
|
|
1183
1289
|
})
|
|
1184
1290
|
|
|
1291
|
+
fetch('/dsh-moa/leaderboard', { cache: 'no-store' })
|
|
1292
|
+
.then((r) => r.json())
|
|
1293
|
+
.then((data) => {
|
|
1294
|
+
if (data && data.ok && Array.isArray(data.models)) {
|
|
1295
|
+
setLeaderboard(data.models.slice(0, 10))
|
|
1296
|
+
}
|
|
1297
|
+
})
|
|
1298
|
+
.catch(() => {})
|
|
1299
|
+
|
|
1185
1300
|
fetch('/dsh-moa/history?limit=100', { cache: 'no-store' })
|
|
1186
1301
|
.then((r) => r.json())
|
|
1187
1302
|
.then((data) => {
|
|
@@ -1290,6 +1405,7 @@ window.__ModuleLoader__.load({
|
|
|
1290
1405
|
setEnabled,
|
|
1291
1406
|
hostStatus,
|
|
1292
1407
|
stats,
|
|
1408
|
+
leaderboard,
|
|
1293
1409
|
handleSave,
|
|
1294
1410
|
saveStatus,
|
|
1295
1411
|
reload,
|
package/lib/index.js
CHANGED
|
@@ -58,6 +58,7 @@ export const PresetSchema = z.object({
|
|
|
58
58
|
aggregator_temperature: z.number().default(0.4),
|
|
59
59
|
reference_timeout_sec: z.number().default(60),
|
|
60
60
|
aggregator_timeout_sec: z.number().default(180),
|
|
61
|
+
blind_evaluation: z.boolean().default(false),
|
|
61
62
|
max_tokens: z.number().default(4096),
|
|
62
63
|
judge_criteria: z.string().default(''),
|
|
63
64
|
})
|
package/lib/moa-prompts.js
CHANGED
|
@@ -33,6 +33,9 @@ Penalize and strictly downgrade candidates exhibiting any of the following flaws
|
|
|
33
33
|
8. 🚫 Context Amnesia & Regressions:
|
|
34
34
|
- Dropping or breaking previously functioning project features while adding new code.`
|
|
35
35
|
|
|
36
|
+
export const LANGUAGE_MIRRORING_DIRECTIVE = `### 🌐 Language Mirroring Requirement
|
|
37
|
+
CRITICAL: You MUST write your entire analysis, reasoning, verdict, explanations and instructions in the EXACT SAME LANGUAGE as the user's prompt (e.g. Russian if the prompt is written in Russian, English if in English). Do NOT switch or translate to English unless explicitly requested by the user.`
|
|
38
|
+
|
|
36
39
|
/**
|
|
37
40
|
* Formats a provider + model slot into a readable string key.
|
|
38
41
|
*/
|
|
@@ -141,15 +144,19 @@ The user gave the task:
|
|
|
141
144
|
The advisors proposed the following decision points and clarifications:
|
|
142
145
|
${joined}
|
|
143
146
|
|
|
147
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
148
|
+
|
|
144
149
|
Your task is to synthesize a single, compact, friendly and structured questionnaire (2-4 questions) in the same language as the user's prompt.
|
|
145
150
|
Each question must offer 2-3 concrete recommended answer options (e.g.: 1. Format: single-file HTML/JS or React? 2. Style: minimalism, iOS or neubrutalism?).
|
|
146
151
|
At the end, add a note that the user can answer briefly (e.g.: "1, 2, dark theme") or trust the defaults.`
|
|
147
152
|
}
|
|
148
153
|
|
|
149
154
|
/**
|
|
150
|
-
* Builds the curator synthesis prompt with component analysis and
|
|
155
|
+
* Builds the curator synthesis prompt with component analysis, model recommendation, and optional blind review.
|
|
151
156
|
*/
|
|
152
|
-
export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '') {
|
|
157
|
+
export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
|
|
158
|
+
const isBlind = Boolean(options.blindEvaluation)
|
|
159
|
+
|
|
153
160
|
const joined = referenceOutputs
|
|
154
161
|
.map((r, i) => {
|
|
155
162
|
const fileSummary = (r.files && r.files.length > 0)
|
|
@@ -159,7 +166,10 @@ export function buildCuratorSynthesisPrompt(userPrompt, referenceOutputs = [], j
|
|
|
159
166
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
160
167
|
textContent = stripOrSummarizeCode(textContent)
|
|
161
168
|
}
|
|
162
|
-
|
|
169
|
+
const header = isBlind
|
|
170
|
+
? `Candidate ${i + 1}:${fileSummary}`
|
|
171
|
+
: `Candidate ${i + 1} — ${r.label}:${fileSummary}`
|
|
172
|
+
return `${header}\n${textContent}`
|
|
163
173
|
})
|
|
164
174
|
.join('\n\n')
|
|
165
175
|
|
|
@@ -178,8 +188,10 @@ ${joined}
|
|
|
178
188
|
|
|
179
189
|
${ANTIPATTERNS_RUBRIC}
|
|
180
190
|
|
|
191
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
192
|
+
|
|
181
193
|
Instructions:
|
|
182
|
-
Your response MUST be structured into three clear parts
|
|
194
|
+
Your response MUST be structured into three clear parts:
|
|
183
195
|
|
|
184
196
|
### 1. 🔍 Curator Analysis & Component Breakdown
|
|
185
197
|
- For EACH candidate, provide:
|
|
@@ -189,7 +201,7 @@ Your response MUST be structured into three clear parts (respond in the same lan
|
|
|
189
201
|
|
|
190
202
|
### 2. 🧩 Assembly Recipe & Recommended Master Assembler
|
|
191
203
|
- Recommend the best single agent model to assemble and finalize the solution:
|
|
192
|
-
RECOMMENDED_ASSEMBLER: <number from 1 to N> (<provider:model>)
|
|
204
|
+
RECOMMENDED_ASSEMBLER: <number from 1 to N> ${isBlind ? '' : '(<provider:model>)'}
|
|
193
205
|
- State the machine winner index marker for file promotion:
|
|
194
206
|
WINNER_CANDIDATE_INDEX: <number from 1 to N>
|
|
195
207
|
- Provide the exact blueprint / instructions for combining the best pieces into a unified deliverable.
|
|
@@ -200,13 +212,15 @@ Your response MUST be structured into three clear parts (respond in the same lan
|
|
|
200
212
|
}
|
|
201
213
|
|
|
202
214
|
/**
|
|
203
|
-
* Builds the judge synthesis prompt for standard or curator mode.
|
|
215
|
+
* Builds the judge synthesis prompt for standard or curator mode with optional blind evaluation.
|
|
204
216
|
*/
|
|
205
217
|
export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCriteria = '', options = {}) {
|
|
206
218
|
if (options.curatorSynthesis) {
|
|
207
|
-
return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria)
|
|
219
|
+
return buildCuratorSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, options)
|
|
208
220
|
}
|
|
209
221
|
|
|
222
|
+
const isBlind = Boolean(options.blindEvaluation)
|
|
223
|
+
|
|
210
224
|
const joined = referenceOutputs
|
|
211
225
|
.map((r, i) => {
|
|
212
226
|
const fileSummary = (r.files && r.files.length > 0)
|
|
@@ -216,7 +230,10 @@ export function buildSynthesisPrompt(userPrompt, referenceOutputs = [], judgeCri
|
|
|
216
230
|
if (referenceOutputs.length >= 3 && textContent.length > 3000) {
|
|
217
231
|
textContent = stripOrSummarizeCode(textContent)
|
|
218
232
|
}
|
|
219
|
-
|
|
233
|
+
const header = isBlind
|
|
234
|
+
? `Reference ${i + 1}:${fileSummary}`
|
|
235
|
+
: `Reference ${i + 1} — ${r.label}:${fileSummary}`
|
|
236
|
+
return `${header}\n${textContent}`
|
|
220
237
|
})
|
|
221
238
|
.join('\n\n')
|
|
222
239
|
|
|
@@ -234,11 +251,13 @@ ${joined}
|
|
|
234
251
|
|
|
235
252
|
${ANTIPATTERNS_RUBRIC}
|
|
236
253
|
|
|
254
|
+
${LANGUAGE_MIRRORING_DIRECTIVE}
|
|
255
|
+
|
|
237
256
|
Instructions:
|
|
238
|
-
Your response MUST be structured into three clear parts
|
|
257
|
+
Your response MUST be structured into three clear parts:
|
|
239
258
|
|
|
240
259
|
### 1. ⚖️ Judge verdict and comparative analysis
|
|
241
|
-
- **Winner**: clearly name the
|
|
260
|
+
- **Winner**: clearly name the chosen candidate (e.g. "Winner: Candidate 1" or "Reference 1 is chosen").
|
|
242
261
|
- Always add the machine winner-selection marker:
|
|
243
262
|
WINNER_CANDIDATE_INDEX: <number from 1 to N>
|
|
244
263
|
- **Why this choice**: compare code, architecture, strengths, weaknesses and reliability of all candidates in detail.
|
package/lib/moa-runner.js
CHANGED
|
@@ -327,6 +327,7 @@ export async function runMoAPipeline({
|
|
|
327
327
|
const candidateRetries = 1
|
|
328
328
|
const refTimeoutSec = typeof preset?.reference_timeout_sec === 'number' ? preset.reference_timeout_sec : 60
|
|
329
329
|
const aggTimeoutSec = typeof preset?.aggregator_timeout_sec === 'number' ? preset.aggregator_timeout_sec : 180
|
|
330
|
+
const isBlindEvaluation = Boolean(preset?.blind_evaluation)
|
|
330
331
|
|
|
331
332
|
// 2. Collect project context for refinement tasks
|
|
332
333
|
const isRefinement = isRefinementTask(userPrompt)
|
|
@@ -531,7 +532,10 @@ export async function runMoAPipeline({
|
|
|
531
532
|
}
|
|
532
533
|
|
|
533
534
|
// 8. Synthesis phase via primary judge or fallback chain
|
|
534
|
-
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
535
|
+
const synthesisPrompt = buildSynthesisPrompt(userPrompt, referenceOutputs, judgeCriteria, {
|
|
536
|
+
curatorSynthesis: isCuratorSynthesis,
|
|
537
|
+
blindEvaluation: isBlindEvaluation,
|
|
538
|
+
})
|
|
535
539
|
let synthesizedText = ''
|
|
536
540
|
let aggUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 }
|
|
537
541
|
let chosenJudge = primaryJudge
|