@goodandready/dsh-moa 0.2.25 → 0.2.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/README.ru.md +30 -0
- package/lib/client.js +248 -0
- package/lib/history.js +11 -0
- package/lib/index.js +41 -0
- package/lib/moa-budget.js +97 -0
- package/lib/moa-candidates.js +90 -22
- package/lib/moa-context.js +70 -0
- package/lib/moa-multi-judge.js +301 -0
- package/lib/moa-prompts.js +25 -6
- package/lib/moa-report.js +124 -0
- package/lib/moa-router.js +102 -0
- package/lib/moa-runner.js +131 -152
- package/lib/moa-stream.js +117 -0
- package/lib/moa-test-gate.js +236 -0
- package/lib/routes.js +58 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -142,6 +142,36 @@ Inspect line-by-line differences between candidate proposals and the curator's s
|
|
|
142
142
|
### 8. Pre-Promotion Git Checkpoints (`dsh-time-machine`)
|
|
143
143
|
Before promoting any winning candidate files over the workspace root, `dsh-moa` invokes the local `dsh-time-machine` service to create a shadow Git checkpoint (`moa-pre-promotion: candidate-N`). If `dsh-time-machine` is absent or unreachable, file promotion proceeds seamlessly via best-effort fallback.
|
|
144
144
|
|
|
145
|
+
### 9. Automated Test Execution Gate
|
|
146
|
+
When `test_gate_enabled: true` and `test_command` (e.g. `npm test` or `pytest`) are configured, candidate code is executed in an ephemeral sandbox overlay. The judge receives concrete test outcomes, durations, and output logs to ground decisions in objective verification.
|
|
147
|
+
|
|
148
|
+
### 10. Multi-Judge Panel & Consensus Voting
|
|
149
|
+
When `multi_judge_enabled: true`, candidate solutions are independently evaluated by a panel of judge models. Winner selection supports `majority`, `highest_score`, or `unanimous` consensus strategies.
|
|
150
|
+
|
|
151
|
+
### 11. Composite Hybrid Synthesis (AST / Block Merge)
|
|
152
|
+
When `composite_merge_enabled: true`, the aggregator assembles a modular hybrid: combining the strongest core logic from one model, robust error handling from another, and complete types/tests from a third.
|
|
153
|
+
|
|
154
|
+
### 12. Smart Dynamic Preset Router & JEV Classifier
|
|
155
|
+
When invoking `/moa` without explicit preset flags, the router classifies prompt intent using keyword heuristics or a zero-shot JEV model (`smart_routing_model`) to automatically select the optimal preset.
|
|
156
|
+
|
|
157
|
+
### 13. Real-Time Live Cost Counter & Streaming Ticker
|
|
158
|
+
As each candidate completes, live token counts and USD costs are streamed directly into the chat based on catalog pricing and vendor rates.
|
|
159
|
+
|
|
160
|
+
### 14. Cost Budget Guardrails (Trim & Abort Modes)
|
|
161
|
+
`budget_guard_enabled` and `max_budget_usd` guard against accidental spend. Mode `trim` automatically reduces the candidate pool to fit the budget, while `abort` cancels execution before token consumption.
|
|
162
|
+
|
|
163
|
+
### 15. Temperature Gradient Exploration & Per-Candidate Temperature
|
|
164
|
+
Supports individual candidate temperatures (`slot.temperature`) and `temperature_gradient_enabled` to distribute temperatures (0.2 → 0.9) across candidates for maximum architectural diversity.
|
|
165
|
+
|
|
166
|
+
### 16. Multi-Turn Conversation Memory & Context Pruning
|
|
167
|
+
Seamlessly retains synthesized baseline code from prior turns while pruning intermediate noise to keep token overhead low.
|
|
168
|
+
|
|
169
|
+
### 17. Automated Benchmark & Post-Mortem PR Reports
|
|
170
|
+
`report_generation_enabled` generates comprehensive Markdown & JSON reports detailing candidate metrics, agreement scores, test gate results, and cost breakdowns via `GET /dsh-moa/runs/:id/report`.
|
|
171
|
+
|
|
172
|
+
### 18. Graceful Degradation & Local Fallback Resilience
|
|
173
|
+
When `local_fallback_enabled: true`, candidates that fail due to network outages or rate limits (429/500) automatically recover using configured local models (Ollama / MiniPC).
|
|
174
|
+
|
|
145
175
|
### 6. Live Canvas 1-Click Preview (optional)
|
|
146
176
|
If `@goodandready/dsh-live-canvas` is installed in the same profile, `dsh-moa` pushes the promoted HTML file to the Live Canvas REST contract (`POST /dsh-live-canvas/api/preview`, served by the same harness webServer) and appends a one-click preview link (`/dsh-live-canvas/sandbox/<id>`) to the answer. Without the plugin the step is skipped silently — no errors in the log, no dead links.
|
|
147
177
|
|
package/README.ru.md
CHANGED
|
@@ -142,6 +142,36 @@ graph TD
|
|
|
142
142
|
### 8. Теневые Git-чекпоинты (`dsh-time-machine`)
|
|
143
143
|
Перед промоушном файлов победителя в корень рабочей области `dsh-moa` автоматически обращается к сервису `dsh-time-machine` для создания снимка (`moa-pre-promotion: candidate-N`). Если плагин недоступен, промоушн штатно продолжается по схеме best-effort.
|
|
144
144
|
|
|
145
|
+
### 9. Автоматический шлюз выполнения тестов (Automated Test Execution Gate)
|
|
146
|
+
При включении `test_gate_enabled: true` и указании тестовой команды (`test_command`, например `npm test` или `pytest`) каждый кандидат запускается в изолированном оверлейном окружении. Судья получает объективный вердикт с кодами возврата, временем исполнения и логами тестов до вынесения итогового решения.
|
|
147
|
+
|
|
148
|
+
### 10. Коллегия судей и консенсусное голосование (Multi-Judge Panel)
|
|
149
|
+
При активации опции `multi_judge_enabled` решения кандидатов оценивает независимая панель судейских моделей с поддержкой стратегий голосования: `majority` (большинство), `highest_score` (наивысший средний балл) или `unanimous` (единогласный консенсус).
|
|
150
|
+
|
|
151
|
+
### 11. Композитный гибридный синтез (Composite AST/Block Merge)
|
|
152
|
+
Опция `composite_merge_enabled` активирует алгоритм объединения модулей: судья не просто выбирает одного кандидата, а синтезирует итоговое решение из лучших компонентов всех участников (чистейшая алгоритмическая база от одного, надежные обработчики ошибок от второго, исчерпывающие типы и тесты от третьего).
|
|
153
|
+
|
|
154
|
+
### 12. Умный динамический роутинг пресетов и модель JEV
|
|
155
|
+
При запуске `/moa` без явного указания флага встроенный классификатор определяет намерение (интерфейсы, безопасность, рефакторинг, баги, архитектура) по эвристическим правилам или с помощью легковесной модели JEV (`smart_routing_model`) и автоматически подбирает оптимальный пресет.
|
|
156
|
+
|
|
157
|
+
### 13. Живой счетчик стоимости и потоковая телеметрия
|
|
158
|
+
По мере завершения каждого кандидата плагин в реальном времени транслирует в чат накопленный расход токенов и расчетную стоимость в USD на базе встроенного каталога тарифов OpenRouter и прямых цен провайдеров.
|
|
159
|
+
|
|
160
|
+
### 14. Бюджетные ограничения расходов (Cost Budget Guardrails)
|
|
161
|
+
Опция `budget_guard_enabled` с лимитом `max_budget_usd` защищает от непредвиденных трат: режим `trim` автоматически сокращает пул кандидатов до самых экономичных моделей, укладывающихся в бюджет, а режим `abort` прерывает выполнение до начала генерации.
|
|
162
|
+
|
|
163
|
+
### 15. Градиент температурной диверсификации и точечная температура
|
|
164
|
+
Поддерживается как индивидуальная настройка температуры для каждого кандидата (`slot.temperature`), так и опция `temperature_gradient_enabled`, автоматически распределяющая температуры кандидатов по шкале от 0.2 до 0.9 для достижения максимального разнообразия инженерных подходов.
|
|
165
|
+
|
|
166
|
+
### 16. Многошаговый диалоговый контекст и умное усечение
|
|
167
|
+
При продолжении беседы MoA извлекает синтезированное решение предыдущего шага в качестве базовой точки отсчета, сохраняя системный контекст и предотвращая раздувание токенов.
|
|
168
|
+
|
|
169
|
+
### 17. Автоматические бенчмарк-отчеты и Post-Mortem PR
|
|
170
|
+
При включении `report_generation_enabled` генерируется подробный аналитический отчет с матрицей моделей, временами задержек, оценками судей, расходами и результатами тестов. Отчет доступен по API `GET /dsh-moa/runs/:id/report?format=md`.
|
|
171
|
+
|
|
172
|
+
### 18. Плавная деградация и локальный фоллбэк
|
|
173
|
+
Опция `local_fallback_enabled` позволяет задать резервные локальные модели (Ollama, vLLM, MiniPC), к которым плагин бесшовно обращается при сбоях сетевых API или превышении лимитов rate limit (429/500).
|
|
174
|
+
|
|
145
175
|
### 6. Live Canvas 1-клик предпросмотр (опционально)
|
|
146
176
|
Если в профиле установлен `@goodandready/dsh-live-canvas`, `dsh-moa` отправляет промоученный HTML-файл в REST-контракт Live Canvas (`POST /dsh-live-canvas/api/preview`, тот же webServer харнесса) и добавляет к ответу ссылку на предпросмотр (`/dsh-live-canvas/sandbox/<id>`). Без плагина шаг пропускается тихо — без ошибок в журнале и без битых ссылок.
|
|
147
177
|
|
package/lib/client.js
CHANGED
|
@@ -63,6 +63,10 @@ window.__ModuleLoader__.load({
|
|
|
63
63
|
'aggregator.peer_hint': 'Candidates critique each other\'s solutions and refine their code before final judge synthesis.',
|
|
64
64
|
'aggregator.override_label': 'Allow Candidate Override Actions',
|
|
65
65
|
'aggregator.override_hint': 'Keeps candidate workspaces in .moa to let you apply any candidate files via chat command.',
|
|
66
|
+
'aggregator.test_gate_label': 'Test Execution Gate (Automatic Sandbox Testing)',
|
|
67
|
+
'aggregator.test_gate_hint': 'Executes test suite inside candidate sandboxes before judge review. Real test results guide winner selection.',
|
|
68
|
+
'aggregator.test_cmd_placeholder': 'e.g. npm test or node --test',
|
|
69
|
+
'aggregator.test_timeout_label': 'Test timeout (sec):',
|
|
66
70
|
'proposers.role_label': 'Persona:',
|
|
67
71
|
'proposers.role_general': 'General',
|
|
68
72
|
'proposers.role_minimalist': 'Minimalist',
|
|
@@ -177,6 +181,10 @@ window.__ModuleLoader__.load({
|
|
|
177
181
|
'aggregator.peer_hint': '在主裁判终审前,各候选模型互相审阅并改进彼此的方案代码。',
|
|
178
182
|
'aggregator.override_label': '允许手动候选方案提拔操作 (Override)',
|
|
179
183
|
'aggregator.override_hint': '在 .moa 目录中保留候选模型的工作区,允许通过聊天命令提拔任意候选模型的文件。',
|
|
184
|
+
'aggregator.test_gate_label': '测试门禁 (沙箱自动化测试)',
|
|
185
|
+
'aggregator.test_gate_hint': '在裁判评审前在候选沙箱中执行测试套件,真实测试结果将指导获胜者评选。',
|
|
186
|
+
'aggregator.test_cmd_placeholder': '例如 npm test 或 node --test',
|
|
187
|
+
'aggregator.test_timeout_label': '测试超时时间 (秒):',
|
|
180
188
|
'proposers.role_label': '工程角色画像:',
|
|
181
189
|
'proposers.role_general': '通用平衡 (General)',
|
|
182
190
|
'proposers.role_minimalist': '极简标准库 (Minimalist)',
|
|
@@ -1273,6 +1281,215 @@ window.__ModuleLoader__.load({
|
|
|
1273
1281
|
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1274
1282
|
t('aggregator.override_hint')
|
|
1275
1283
|
),
|
|
1284
|
+
React.createElement(
|
|
1285
|
+
'label',
|
|
1286
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 8 } },
|
|
1287
|
+
React.createElement('input', {
|
|
1288
|
+
type: 'checkbox',
|
|
1289
|
+
checked: Boolean(currentPreset.test_gate_enabled),
|
|
1290
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, test_gate_enabled: e.target.checked })),
|
|
1291
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1292
|
+
}),
|
|
1293
|
+
t('aggregator.test_gate_label')
|
|
1294
|
+
),
|
|
1295
|
+
React.createElement(
|
|
1296
|
+
'div',
|
|
1297
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1298
|
+
t('aggregator.test_gate_hint')
|
|
1299
|
+
),
|
|
1300
|
+
Boolean(currentPreset.test_gate_enabled) &&
|
|
1301
|
+
React.createElement(
|
|
1302
|
+
'div',
|
|
1303
|
+
{ style: { marginLeft: 22, marginTop: 6, display: 'flex', flexDirection: 'column', gap: 6 } },
|
|
1304
|
+
React.createElement('input', {
|
|
1305
|
+
type: 'text',
|
|
1306
|
+
className: 'moa-input',
|
|
1307
|
+
style: { fontSize: 12, height: 28 },
|
|
1308
|
+
placeholder: t('aggregator.test_cmd_placeholder'),
|
|
1309
|
+
value: currentPreset.test_command || '',
|
|
1310
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, test_command: e.target.value })),
|
|
1311
|
+
}),
|
|
1312
|
+
React.createElement(
|
|
1313
|
+
'div',
|
|
1314
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8 } },
|
|
1315
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('aggregator.test_timeout_label')),
|
|
1316
|
+
React.createElement('input', {
|
|
1317
|
+
type: 'number',
|
|
1318
|
+
className: 'moa-input',
|
|
1319
|
+
style: { width: 70, height: 26, fontSize: 12 },
|
|
1320
|
+
min: 5,
|
|
1321
|
+
max: 120,
|
|
1322
|
+
value: currentPreset.test_gate_timeout_sec ?? 15,
|
|
1323
|
+
onChange: (e) => {
|
|
1324
|
+
const val = parseInt(e.target.value, 10)
|
|
1325
|
+
updateCurrentPreset((p) => ({ ...p, test_gate_timeout_sec: isNaN(val) ? 15 : val }))
|
|
1326
|
+
},
|
|
1327
|
+
})
|
|
1328
|
+
)
|
|
1329
|
+
),
|
|
1330
|
+
/* Multi-Judge Panel & Consensus Voting */
|
|
1331
|
+
React.createElement(
|
|
1332
|
+
'div',
|
|
1333
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1334
|
+
React.createElement(
|
|
1335
|
+
'label',
|
|
1336
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1337
|
+
React.createElement('input', {
|
|
1338
|
+
type: 'checkbox',
|
|
1339
|
+
checked: Boolean(currentPreset.multi_judge_enabled),
|
|
1340
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, multi_judge_enabled: e.target.checked })),
|
|
1341
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1342
|
+
}),
|
|
1343
|
+
'Multi-Judge Panel & Consensus Voting'
|
|
1344
|
+
),
|
|
1345
|
+
React.createElement(
|
|
1346
|
+
'div',
|
|
1347
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1348
|
+
'Deploy multiple judge models to score candidates and determine winner via consensus voting.'
|
|
1349
|
+
),
|
|
1350
|
+
currentPreset.multi_judge_enabled && React.createElement(
|
|
1351
|
+
'div',
|
|
1352
|
+
{ style: { marginLeft: 22, display: 'flex', alignItems: 'center', gap: 8, marginTop: 4 } },
|
|
1353
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Voting Strategy:'),
|
|
1354
|
+
React.createElement(
|
|
1355
|
+
'select',
|
|
1356
|
+
{
|
|
1357
|
+
className: 'moa-select',
|
|
1358
|
+
style: { width: 140, height: 28, fontSize: 12, padding: '0 6px' },
|
|
1359
|
+
value: currentPreset.judge_voting_strategy || 'majority',
|
|
1360
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, judge_voting_strategy: e.target.value })),
|
|
1361
|
+
},
|
|
1362
|
+
React.createElement('option', { value: 'majority' }, 'Majority Vote'),
|
|
1363
|
+
React.createElement('option', { value: 'highest_score' }, 'Highest Score'),
|
|
1364
|
+
React.createElement('option', { value: 'unanimous' }, 'Unanimous')
|
|
1365
|
+
)
|
|
1366
|
+
)
|
|
1367
|
+
),
|
|
1368
|
+
|
|
1369
|
+
/* Composite Hybrid Synthesis (AST / Block Merge) */
|
|
1370
|
+
React.createElement(
|
|
1371
|
+
'div',
|
|
1372
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1373
|
+
React.createElement(
|
|
1374
|
+
'label',
|
|
1375
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1376
|
+
React.createElement('input', {
|
|
1377
|
+
type: 'checkbox',
|
|
1378
|
+
checked: Boolean(currentPreset.composite_merge_enabled),
|
|
1379
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, composite_merge_enabled: e.target.checked })),
|
|
1380
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1381
|
+
}),
|
|
1382
|
+
'Composite Hybrid Synthesis (AST / Block Merge)'
|
|
1383
|
+
),
|
|
1384
|
+
React.createElement(
|
|
1385
|
+
'div',
|
|
1386
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1387
|
+
'Synthesize modular code blocks, types, and algorithms from multiple candidates into a unified composite.'
|
|
1388
|
+
)
|
|
1389
|
+
),
|
|
1390
|
+
|
|
1391
|
+
/* Cost Budget Guardrails & Auto-Fallback */
|
|
1392
|
+
React.createElement(
|
|
1393
|
+
'div',
|
|
1394
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1395
|
+
React.createElement(
|
|
1396
|
+
'label',
|
|
1397
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1398
|
+
React.createElement('input', {
|
|
1399
|
+
type: 'checkbox',
|
|
1400
|
+
checked: Boolean(currentPreset.budget_guard_enabled),
|
|
1401
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, budget_guard_enabled: e.target.checked })),
|
|
1402
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1403
|
+
}),
|
|
1404
|
+
'Cost Budget Guardrails & Auto-Fallback'
|
|
1405
|
+
),
|
|
1406
|
+
React.createElement(
|
|
1407
|
+
'div',
|
|
1408
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1409
|
+
'Cap maximum run cost; auto-trim candidate pool or abort to prevent accidental token spend.'
|
|
1410
|
+
),
|
|
1411
|
+
currentPreset.budget_guard_enabled && React.createElement(
|
|
1412
|
+
'div',
|
|
1413
|
+
{ style: { marginLeft: 22, display: 'flex', alignItems: 'center', gap: 12, marginTop: 4 } },
|
|
1414
|
+
React.createElement(
|
|
1415
|
+
'div',
|
|
1416
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 6 } },
|
|
1417
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Max USD ($):'),
|
|
1418
|
+
React.createElement('input', {
|
|
1419
|
+
type: 'number',
|
|
1420
|
+
step: '0.01',
|
|
1421
|
+
className: 'moa-input',
|
|
1422
|
+
style: { width: 80, height: 28, padding: '0 6px', fontSize: 12 },
|
|
1423
|
+
value: currentPreset.max_budget_usd ?? 0.1,
|
|
1424
|
+
onChange: (e) => {
|
|
1425
|
+
const val = parseFloat(e.target.value)
|
|
1426
|
+
updateCurrentPreset((p) => ({ ...p, max_budget_usd: isNaN(val) ? 0 : val }))
|
|
1427
|
+
},
|
|
1428
|
+
})
|
|
1429
|
+
),
|
|
1430
|
+
React.createElement(
|
|
1431
|
+
'div',
|
|
1432
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 6 } },
|
|
1433
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Action:'),
|
|
1434
|
+
React.createElement(
|
|
1435
|
+
'select',
|
|
1436
|
+
{
|
|
1437
|
+
className: 'moa-select',
|
|
1438
|
+
style: { width: 100, height: 28, fontSize: 12, padding: '0 6px' },
|
|
1439
|
+
value: currentPreset.budget_action || 'trim',
|
|
1440
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, budget_action: e.target.value })),
|
|
1441
|
+
},
|
|
1442
|
+
React.createElement('option', { value: 'trim' }, 'Trim Pool'),
|
|
1443
|
+
React.createElement('option', { value: 'abort' }, 'Abort')
|
|
1444
|
+
)
|
|
1445
|
+
)
|
|
1446
|
+
)
|
|
1447
|
+
),
|
|
1448
|
+
|
|
1449
|
+
/* Automated Benchmark & Post-Mortem Report */
|
|
1450
|
+
React.createElement(
|
|
1451
|
+
'div',
|
|
1452
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1453
|
+
React.createElement(
|
|
1454
|
+
'label',
|
|
1455
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1456
|
+
React.createElement('input', {
|
|
1457
|
+
type: 'checkbox',
|
|
1458
|
+
checked: Boolean(currentPreset.report_generation_enabled),
|
|
1459
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, report_generation_enabled: e.target.checked })),
|
|
1460
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1461
|
+
}),
|
|
1462
|
+
'Automated Benchmark & Post-Mortem Report'
|
|
1463
|
+
),
|
|
1464
|
+
React.createElement(
|
|
1465
|
+
'div',
|
|
1466
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1467
|
+
'Generate downloadable Markdown & JSON reports detailing candidate latency, agreement, test gate, and cost breakdown.'
|
|
1468
|
+
)
|
|
1469
|
+
),
|
|
1470
|
+
|
|
1471
|
+
/* Graceful Degradation & Local Fallback */
|
|
1472
|
+
React.createElement(
|
|
1473
|
+
'div',
|
|
1474
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1475
|
+
React.createElement(
|
|
1476
|
+
'label',
|
|
1477
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1478
|
+
React.createElement('input', {
|
|
1479
|
+
type: 'checkbox',
|
|
1480
|
+
checked: Boolean(currentPreset.local_fallback_enabled),
|
|
1481
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, local_fallback_enabled: e.target.checked })),
|
|
1482
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1483
|
+
}),
|
|
1484
|
+
'Graceful Degradation & Local Fallback'
|
|
1485
|
+
),
|
|
1486
|
+
React.createElement(
|
|
1487
|
+
'div',
|
|
1488
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1489
|
+
'Automatically fall back to local Ollama / MiniPC models when online candidate APIs fail.'
|
|
1490
|
+
)
|
|
1491
|
+
),
|
|
1492
|
+
|
|
1276
1493
|
/* Fallback Judges Chain */
|
|
1277
1494
|
React.createElement(
|
|
1278
1495
|
'div',
|
|
@@ -1427,6 +1644,17 @@ window.__ModuleLoader__.load({
|
|
|
1427
1644
|
updateCurrentPreset((p) => ({ ...p, reference_timeout_sec: isNaN(val) ? 60 : val }))
|
|
1428
1645
|
},
|
|
1429
1646
|
})
|
|
1647
|
+
),
|
|
1648
|
+
React.createElement(
|
|
1649
|
+
'label',
|
|
1650
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 4 } },
|
|
1651
|
+
React.createElement('input', {
|
|
1652
|
+
type: 'checkbox',
|
|
1653
|
+
checked: Boolean(currentPreset.temperature_gradient_enabled),
|
|
1654
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, temperature_gradient_enabled: e.target.checked })),
|
|
1655
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1656
|
+
}),
|
|
1657
|
+
'Temperature Gradient Exploration (0.2 → 0.9 across candidates)'
|
|
1430
1658
|
)
|
|
1431
1659
|
),
|
|
1432
1660
|
React.createElement(
|
|
@@ -1448,6 +1676,26 @@ window.__ModuleLoader__.load({
|
|
|
1448
1676
|
t,
|
|
1449
1677
|
})
|
|
1450
1678
|
),
|
|
1679
|
+
React.createElement('input', {
|
|
1680
|
+
type: 'number',
|
|
1681
|
+
step: '0.1',
|
|
1682
|
+
min: '0',
|
|
1683
|
+
max: '2',
|
|
1684
|
+
placeholder: 'Temp',
|
|
1685
|
+
title: 'Candidate Temperature (leave empty for default/gradient)',
|
|
1686
|
+
className: 'moa-input',
|
|
1687
|
+
style: { width: 50, height: 32, fontSize: 12, padding: '0 4px', textAlign: 'center' },
|
|
1688
|
+
value: (typeof ref.temperature === 'number' && ref.temperature >= 0) ? ref.temperature : '',
|
|
1689
|
+
onChange: (e) => {
|
|
1690
|
+
const raw = e.target.value
|
|
1691
|
+
const parsedVal = raw === '' ? -1 : parseFloat(raw)
|
|
1692
|
+
updateCurrentPreset((p) => {
|
|
1693
|
+
const refs = [...(p.reference_models || [])]
|
|
1694
|
+
refs[idx] = { ...refs[idx], temperature: isNaN(parsedVal) ? -1 : parsedVal }
|
|
1695
|
+
return { ...p, reference_models: refs }
|
|
1696
|
+
})
|
|
1697
|
+
},
|
|
1698
|
+
}),
|
|
1451
1699
|
React.createElement(
|
|
1452
1700
|
'select',
|
|
1453
1701
|
{
|
package/lib/history.js
CHANGED
|
@@ -86,6 +86,8 @@ function normalizeRunRecord(record) {
|
|
|
86
86
|
totalTokens: Number(record.totalTokens || 0),
|
|
87
87
|
totalCostUsd: Number(record.totalCostUsd || 0),
|
|
88
88
|
durationMs: Number(record.durationMs || 0),
|
|
89
|
+
benchmarkReport: record.benchmarkReport || null,
|
|
90
|
+
consensus: record.consensus || null,
|
|
89
91
|
}
|
|
90
92
|
}
|
|
91
93
|
|
|
@@ -413,8 +415,17 @@ export function candidatesForHistory(referenceOutputs) {
|
|
|
413
415
|
provider: r.slot?.provider || "",
|
|
414
416
|
model: r.slot?.model || "",
|
|
415
417
|
role: r.role_persona || r.slot?.role_persona || r.slot?.role || "general",
|
|
418
|
+
temperature: r.temperature,
|
|
419
|
+
wasFallback: Boolean(r.was_fallback),
|
|
420
|
+
originalModel: r.original_model || null,
|
|
416
421
|
files: (r.files || []).map((f) => f.relativePath),
|
|
417
422
|
usage: r.usage || { inputTokens: 0, outputTokens: 0 },
|
|
418
423
|
costUsd: r.costUsd || 0,
|
|
424
|
+
testResult: r.testResult ? {
|
|
425
|
+
passed: r.testResult.passed,
|
|
426
|
+
exitCode: r.testResult.exitCode,
|
|
427
|
+
summary: r.testResult.summary,
|
|
428
|
+
durationMs: r.testResult.durationMs,
|
|
429
|
+
} : null,
|
|
419
430
|
}))
|
|
420
431
|
}
|
package/lib/index.js
CHANGED
|
@@ -32,6 +32,7 @@ import {
|
|
|
32
32
|
} from './moa-runner.js'
|
|
33
33
|
|
|
34
34
|
import { registerMoaRoutes } from './routes.js'
|
|
35
|
+
import { resolvePresetForPrompt } from './moa-router.js'
|
|
35
36
|
|
|
36
37
|
import { createLiveCanvasClient } from './live-canvas.js'
|
|
37
38
|
import { registerPluginUpdater, isSafeWriteRequest } from './updater.js'
|
|
@@ -52,6 +53,7 @@ export const ModelSlotSchema = z.object({
|
|
|
52
53
|
provider: z.string().default('opencode-go'),
|
|
53
54
|
model: z.string().default('deepseek-v4-flash'),
|
|
54
55
|
role_persona: z.string().default(''),
|
|
56
|
+
temperature: z.number().default(-1),
|
|
55
57
|
})
|
|
56
58
|
|
|
57
59
|
export const PresetSchema = z.object({
|
|
@@ -74,11 +76,30 @@ export const PresetSchema = z.object({
|
|
|
74
76
|
allow_candidate_override: z.boolean().default(false),
|
|
75
77
|
max_tokens: z.number().default(4096),
|
|
76
78
|
judge_criteria: z.string().default(''),
|
|
79
|
+
test_gate_enabled: z.boolean().default(false),
|
|
80
|
+
test_command: z.string().default(''),
|
|
81
|
+
test_gate_timeout_sec: z.number().default(15),
|
|
82
|
+
multi_judge_enabled: z.boolean().default(false),
|
|
83
|
+
judge_models: z.array(ModelSlotSchema).default([]),
|
|
84
|
+
judge_voting_strategy: z.string().default('majority'),
|
|
85
|
+
composite_merge_enabled: z.boolean().default(false),
|
|
86
|
+
smart_routing_enabled: z.boolean().default(false),
|
|
87
|
+
smart_routing_model: ModelSlotSchema.default({ provider: 'opencode-go', model: 'jev' }),
|
|
88
|
+
budget_guard_enabled: z.boolean().default(false),
|
|
89
|
+
max_budget_usd: z.number().default(0),
|
|
90
|
+
budget_action: z.string().default('trim'),
|
|
91
|
+
temperature_gradient_enabled: z.boolean().default(false),
|
|
92
|
+
multi_turn_enabled: z.boolean().default(true),
|
|
93
|
+
report_generation_enabled: z.boolean().default(false),
|
|
94
|
+
local_fallback_enabled: z.boolean().default(false),
|
|
95
|
+
local_fallback_models: z.array(ModelSlotSchema).default([]),
|
|
77
96
|
})
|
|
78
97
|
|
|
79
98
|
export const Config = z.object({
|
|
80
99
|
enabled: z.boolean().default(true).volatile(),
|
|
81
100
|
default_preset: z.string().default('default').volatile(),
|
|
101
|
+
smart_routing_enabled: z.boolean().default(false).volatile(),
|
|
102
|
+
smart_routing_model: ModelSlotSchema.default({ provider: 'opencode-go', model: 'jev' }).volatile(),
|
|
82
103
|
prices: z.dict(PriceRow).default({}).volatile(),
|
|
83
104
|
presets: z.array(PresetSchema).default([
|
|
84
105
|
{
|
|
@@ -335,6 +356,24 @@ export function apply(ctx, config) {
|
|
|
335
356
|
const cfg = live()
|
|
336
357
|
if (cfg.enabled !== false) {
|
|
337
358
|
const parsed = parseMoACommand(userText, cfg.presets || []) || { prompt: userText, presetName: cfg.default_preset || 'default' }
|
|
359
|
+
const isExplicitPreset = userText.includes('--preset') || userText.includes('-p ')
|
|
360
|
+
if (!isExplicitPreset && (cfg.smart_routing_enabled || cfg.presets?.some((p) => p.smart_routing_enabled))) {
|
|
361
|
+
try {
|
|
362
|
+
const routed = await resolvePresetForPrompt({
|
|
363
|
+
prompt: parsed.prompt || userText,
|
|
364
|
+
presets: cfg.presets || [],
|
|
365
|
+
defaultPreset: parsed.presetName || cfg.default_preset || 'default',
|
|
366
|
+
routingModel: cfg.smart_routing_model,
|
|
367
|
+
callLlm,
|
|
368
|
+
enabled: true,
|
|
369
|
+
})
|
|
370
|
+
if (routed?.isAutoRouted && routed.presetName) {
|
|
371
|
+
parsed.presetName = routed.presetName
|
|
372
|
+
}
|
|
373
|
+
} catch {
|
|
374
|
+
// fallback silently
|
|
375
|
+
}
|
|
376
|
+
}
|
|
338
377
|
const targetPreset = (cfg.presets || []).find((p) => p.name === parsed.presetName) || cfg.presets?.[0]
|
|
339
378
|
const aggProvider = targetPreset?.aggregator?.provider || 'codex'
|
|
340
379
|
const aggModel = targetPreset?.aggregator?.model || 'gpt-5.6-sol'
|
|
@@ -394,3 +433,5 @@ export function apply(ctx, config) {
|
|
|
394
433
|
})
|
|
395
434
|
}, 'dsh-moa: llm stream interceptor')
|
|
396
435
|
}
|
|
436
|
+
|
|
437
|
+
export { executeTestGateForCandidates, runCandidateTestGate } from './moa-test-gate.js'
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cost Budget Guardrails & Auto-Trimming
|
|
3
|
+
* Feature 6 (Cost Budget Guardrails)
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import { estimateTokenCost } from './pricing.js'
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Calculates conservative upfront cost estimate for a list of candidate slots.
|
|
10
|
+
*/
|
|
11
|
+
export function estimateCandidateRunCost(slots = [], prices = {}, promptTokens = 500, expectedOutputTokens = 2048) {
|
|
12
|
+
let totalUsd = 0
|
|
13
|
+
const slotEstimates = []
|
|
14
|
+
|
|
15
|
+
for (const slot of slots) {
|
|
16
|
+
const costInfo = estimateTokenCost(
|
|
17
|
+
slot,
|
|
18
|
+
{ inputTokens: promptTokens, outputTokens: expectedOutputTokens },
|
|
19
|
+
prices
|
|
20
|
+
)
|
|
21
|
+
const cost = costInfo?.costUsd || 0
|
|
22
|
+
totalUsd += cost
|
|
23
|
+
slotEstimates.push({ slot, costUsd: cost })
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
return {
|
|
27
|
+
totalEstimatedUsd: Number(totalUsd.toFixed(6)),
|
|
28
|
+
slotEstimates,
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Applies budget guardrails before initiating parallel candidate execution.
|
|
34
|
+
*/
|
|
35
|
+
export function applyBudgetGuardrails({
|
|
36
|
+
references = [],
|
|
37
|
+
enabled = false,
|
|
38
|
+
maxBudgetUsd = 0,
|
|
39
|
+
action = 'trim', // 'trim' | 'abort'
|
|
40
|
+
prices = {},
|
|
41
|
+
promptLength = 1000,
|
|
42
|
+
}) {
|
|
43
|
+
if (!enabled || typeof maxBudgetUsd !== 'number' || maxBudgetUsd <= 0 || references.length === 0) {
|
|
44
|
+
return {
|
|
45
|
+
allowed: true,
|
|
46
|
+
references,
|
|
47
|
+
action: 'none',
|
|
48
|
+
estimatedCostUsd: 0,
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const promptTokens = Math.max(50, Math.round(promptLength / 4))
|
|
53
|
+
const { totalEstimatedUsd, slotEstimates } = estimateCandidateRunCost(references, prices, promptTokens)
|
|
54
|
+
|
|
55
|
+
if (totalEstimatedUsd <= maxBudgetUsd) {
|
|
56
|
+
return {
|
|
57
|
+
allowed: true,
|
|
58
|
+
references,
|
|
59
|
+
action: 'pass',
|
|
60
|
+
estimatedCostUsd: totalEstimatedUsd,
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
if (action === 'abort') {
|
|
65
|
+
return {
|
|
66
|
+
allowed: false,
|
|
67
|
+
references: [],
|
|
68
|
+
action: 'abort',
|
|
69
|
+
estimatedCostUsd: totalEstimatedUsd,
|
|
70
|
+
maxBudgetUsd,
|
|
71
|
+
reason: `Estimated MoA run cost ($${totalEstimatedUsd.toFixed(4)}) exceeds configured max budget ($${maxBudgetUsd.toFixed(4)}).`,
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Action is 'trim': Sort by estimated cost ascending (cheapest first)
|
|
76
|
+
const sorted = [...slotEstimates].sort((a, b) => a.costUsd - b.costUsd)
|
|
77
|
+
const kept = []
|
|
78
|
+
let cumulative = 0
|
|
79
|
+
|
|
80
|
+
for (const item of sorted) {
|
|
81
|
+
if (cumulative + item.costUsd <= maxBudgetUsd || kept.length === 0) {
|
|
82
|
+
kept.push(item.slot)
|
|
83
|
+
cumulative += item.costUsd
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
allowed: true,
|
|
89
|
+
references: kept,
|
|
90
|
+
action: 'trim',
|
|
91
|
+
originalCount: references.length,
|
|
92
|
+
trimmedCount: kept.length,
|
|
93
|
+
estimatedCostUsd: Number(cumulative.toFixed(6)),
|
|
94
|
+
maxBudgetUsd,
|
|
95
|
+
reason: `Trimmed candidate pool from ${references.length} to ${kept.length} models to stay within budget ($${maxBudgetUsd.toFixed(4)}).`,
|
|
96
|
+
}
|
|
97
|
+
}
|