@goodandready/dsh-moa 0.2.26 → 0.2.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/README.ru.md +30 -0
- package/lib/client.js +194 -0
- package/lib/history.js +5 -0
- package/lib/index.js +36 -0
- package/lib/moa-budget.js +97 -0
- package/lib/moa-candidates.js +90 -22
- package/lib/moa-context.js +70 -0
- package/lib/moa-multi-judge.js +301 -0
- package/lib/moa-prompts.js +3 -0
- package/lib/moa-report.js +124 -0
- package/lib/moa-router.js +102 -0
- package/lib/moa-runner.js +98 -118
- package/lib/moa-stream.js +117 -0
- package/lib/routes.js +58 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -142,6 +142,36 @@ Inspect line-by-line differences between candidate proposals and the curator's s
|
|
|
142
142
|
### 8. Pre-Promotion Git Checkpoints (`dsh-time-machine`)
|
|
143
143
|
Before promoting any winning candidate files over the workspace root, `dsh-moa` invokes the local `dsh-time-machine` service to create a shadow Git checkpoint (`moa-pre-promotion: candidate-N`). If `dsh-time-machine` is absent or unreachable, file promotion proceeds seamlessly via best-effort fallback.
|
|
144
144
|
|
|
145
|
+
### 9. Automated Test Execution Gate
|
|
146
|
+
When `test_gate_enabled: true` and `test_command` (e.g. `npm test` or `pytest`) are configured, candidate code is executed in an ephemeral sandbox overlay. The judge receives concrete test outcomes, durations, and output logs to ground decisions in objective verification.
|
|
147
|
+
|
|
148
|
+
### 10. Multi-Judge Panel & Consensus Voting
|
|
149
|
+
When `multi_judge_enabled: true`, candidate solutions are independently evaluated by a panel of judge models. Winner selection supports `majority`, `highest_score`, or `unanimous` consensus strategies.
|
|
150
|
+
|
|
151
|
+
### 11. Composite Hybrid Synthesis (AST / Block Merge)
|
|
152
|
+
When `composite_merge_enabled: true`, the aggregator assembles a modular hybrid: combining the strongest core logic from one model, robust error handling from another, and complete types/tests from a third.
|
|
153
|
+
|
|
154
|
+
### 12. Smart Dynamic Preset Router & JEV Classifier
|
|
155
|
+
When invoking `/moa` without explicit preset flags, the router classifies prompt intent using keyword heuristics or a zero-shot JEV model (`smart_routing_model`) to automatically select the optimal preset.
|
|
156
|
+
|
|
157
|
+
### 13. Real-Time Live Cost Counter & Streaming Ticker
|
|
158
|
+
As each candidate completes, live token counts and USD costs are streamed directly into the chat based on catalog pricing and vendor rates.
|
|
159
|
+
|
|
160
|
+
### 14. Cost Budget Guardrails (Trim & Abort Modes)
|
|
161
|
+
`budget_guard_enabled` and `max_budget_usd` guard against accidental spend. Mode `trim` automatically reduces the candidate pool to fit the budget, while `abort` cancels execution before token consumption.
|
|
162
|
+
|
|
163
|
+
### 15. Temperature Gradient Exploration & Per-Candidate Temperature
|
|
164
|
+
Supports individual candidate temperatures (`slot.temperature`) and `temperature_gradient_enabled` to distribute temperatures (0.2 → 0.9) across candidates for maximum architectural diversity.
|
|
165
|
+
|
|
166
|
+
### 16. Multi-Turn Conversation Memory & Context Pruning
|
|
167
|
+
Seamlessly retains synthesized baseline code from prior turns while pruning intermediate noise to keep token overhead low.
|
|
168
|
+
|
|
169
|
+
### 17. Automated Benchmark & Post-Mortem PR Reports
|
|
170
|
+
`report_generation_enabled` generates comprehensive Markdown & JSON reports detailing candidate metrics, agreement scores, test gate results, and cost breakdowns via `GET /dsh-moa/runs/:id/report`.
|
|
171
|
+
|
|
172
|
+
### 18. Graceful Degradation & Local Fallback Resilience
|
|
173
|
+
When `local_fallback_enabled: true`, candidates that fail due to network outages or rate limits (429/500) automatically recover using configured local models (Ollama / MiniPC).
|
|
174
|
+
|
|
145
175
|
### 6. Live Canvas 1-Click Preview (optional)
|
|
146
176
|
If `@goodandready/dsh-live-canvas` is installed in the same profile, `dsh-moa` pushes the promoted HTML file to the Live Canvas REST contract (`POST /dsh-live-canvas/api/preview`, served by the same harness webServer) and appends a one-click preview link (`/dsh-live-canvas/sandbox/<id>`) to the answer. Without the plugin the step is skipped silently — no errors in the log, no dead links.
|
|
147
177
|
|
package/README.ru.md
CHANGED
|
@@ -142,6 +142,36 @@ graph TD
|
|
|
142
142
|
### 8. Теневые Git-чекпоинты (`dsh-time-machine`)
|
|
143
143
|
Перед промоушном файлов победителя в корень рабочей области `dsh-moa` автоматически обращается к сервису `dsh-time-machine` для создания снимка (`moa-pre-promotion: candidate-N`). Если плагин недоступен, промоушн штатно продолжается по схеме best-effort.
|
|
144
144
|
|
|
145
|
+
### 9. Автоматический шлюз выполнения тестов (Automated Test Execution Gate)
|
|
146
|
+
При включении `test_gate_enabled: true` и указании тестовой команды (`test_command`, например `npm test` или `pytest`) каждый кандидат запускается в изолированном оверлейном окружении. Судья получает объективный вердикт с кодами возврата, временем исполнения и логами тестов до вынесения итогового решения.
|
|
147
|
+
|
|
148
|
+
### 10. Коллегия судей и консенсусное голосование (Multi-Judge Panel)
|
|
149
|
+
При активации опции `multi_judge_enabled` решения кандидатов оценивает независимая панель судейских моделей с поддержкой стратегий голосования: `majority` (большинство), `highest_score` (наивысший средний балл) или `unanimous` (единогласный консенсус).
|
|
150
|
+
|
|
151
|
+
### 11. Композитный гибридный синтез (Composite AST/Block Merge)
|
|
152
|
+
Опция `composite_merge_enabled` активирует алгоритм объединения модулей: судья не просто выбирает одного кандидата, а синтезирует итоговое решение из лучших компонентов всех участников (чистейшая алгоритмическая база от одного, надежные обработчики ошибок от второго, исчерпывающие типы и тесты от третьего).
|
|
153
|
+
|
|
154
|
+
### 12. Умный динамический роутинг пресетов и модель JEV
|
|
155
|
+
При запуске `/moa` без явного указания флага встроенный классификатор определяет намерение (интерфейсы, безопасность, рефакторинг, баги, архитектура) по эвристическим правилам или с помощью легковесной модели JEV (`smart_routing_model`) и автоматически подбирает оптимальный пресет.
|
|
156
|
+
|
|
157
|
+
### 13. Живой счетчик стоимости и потоковая телеметрия
|
|
158
|
+
По мере завершения каждого кандидата плагин в реальном времени транслирует в чат накопленный расход токенов и расчетную стоимость в USD на базе встроенного каталога тарифов OpenRouter и прямых цен провайдеров.
|
|
159
|
+
|
|
160
|
+
### 14. Бюджетные ограничения расходов (Cost Budget Guardrails)
|
|
161
|
+
Опция `budget_guard_enabled` с лимитом `max_budget_usd` защищает от непредвиденных трат: режим `trim` автоматически сокращает пул кандидатов до самых экономичных моделей, укладывающихся в бюджет, а режим `abort` прерывает выполнение до начала генерации.
|
|
162
|
+
|
|
163
|
+
### 15. Градиент температурной диверсификации и точечная температура
|
|
164
|
+
Поддерживается как индивидуальная настройка температуры для каждого кандидата (`slot.temperature`), так и опция `temperature_gradient_enabled`, автоматически распределяющая температуры кандидатов по шкале от 0.2 до 0.9 для достижения максимального разнообразия инженерных подходов.
|
|
165
|
+
|
|
166
|
+
### 16. Многошаговый диалоговый контекст и умное усечение
|
|
167
|
+
При продолжении беседы MoA извлекает синтезированное решение предыдущего шага в качестве базовой точки отсчета, сохраняя системный контекст и предотвращая раздувание токенов.
|
|
168
|
+
|
|
169
|
+
### 17. Автоматические бенчмарк-отчеты и Post-Mortem PR
|
|
170
|
+
При включении `report_generation_enabled` генерируется подробный аналитический отчет с матрицей моделей, временами задержек, оценками судей, расходами и результатами тестов. Отчет доступен по API `GET /dsh-moa/runs/:id/report?format=md`.
|
|
171
|
+
|
|
172
|
+
### 18. Плавная деградация и локальный фоллбэк
|
|
173
|
+
Опция `local_fallback_enabled` позволяет задать резервные локальные модели (Ollama, vLLM, MiniPC), к которым плагин бесшовно обращается при сбоях сетевых API или превышении лимитов rate limit (429/500).
|
|
174
|
+
|
|
145
175
|
### 6. Live Canvas 1-клик предпросмотр (опционально)
|
|
146
176
|
Если в профиле установлен `@goodandready/dsh-live-canvas`, `dsh-moa` отправляет промоученный HTML-файл в REST-контракт Live Canvas (`POST /dsh-live-canvas/api/preview`, тот же webServer харнесса) и добавляет к ответу ссылку на предпросмотр (`/dsh-live-canvas/sandbox/<id>`). Без плагина шаг пропускается тихо — без ошибок в журнале и без битых ссылок.
|
|
147
177
|
|
package/lib/client.js
CHANGED
|
@@ -1327,6 +1327,169 @@ window.__ModuleLoader__.load({
|
|
|
1327
1327
|
})
|
|
1328
1328
|
)
|
|
1329
1329
|
),
|
|
1330
|
+
/* Multi-Judge Panel & Consensus Voting */
|
|
1331
|
+
React.createElement(
|
|
1332
|
+
'div',
|
|
1333
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1334
|
+
React.createElement(
|
|
1335
|
+
'label',
|
|
1336
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1337
|
+
React.createElement('input', {
|
|
1338
|
+
type: 'checkbox',
|
|
1339
|
+
checked: Boolean(currentPreset.multi_judge_enabled),
|
|
1340
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, multi_judge_enabled: e.target.checked })),
|
|
1341
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1342
|
+
}),
|
|
1343
|
+
'Multi-Judge Panel & Consensus Voting'
|
|
1344
|
+
),
|
|
1345
|
+
React.createElement(
|
|
1346
|
+
'div',
|
|
1347
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1348
|
+
'Deploy multiple judge models to score candidates and determine winner via consensus voting.'
|
|
1349
|
+
),
|
|
1350
|
+
currentPreset.multi_judge_enabled && React.createElement(
|
|
1351
|
+
'div',
|
|
1352
|
+
{ style: { marginLeft: 22, display: 'flex', alignItems: 'center', gap: 8, marginTop: 4 } },
|
|
1353
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Voting Strategy:'),
|
|
1354
|
+
React.createElement(
|
|
1355
|
+
'select',
|
|
1356
|
+
{
|
|
1357
|
+
className: 'moa-select',
|
|
1358
|
+
style: { width: 140, height: 28, fontSize: 12, padding: '0 6px' },
|
|
1359
|
+
value: currentPreset.judge_voting_strategy || 'majority',
|
|
1360
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, judge_voting_strategy: e.target.value })),
|
|
1361
|
+
},
|
|
1362
|
+
React.createElement('option', { value: 'majority' }, 'Majority Vote'),
|
|
1363
|
+
React.createElement('option', { value: 'highest_score' }, 'Highest Score'),
|
|
1364
|
+
React.createElement('option', { value: 'unanimous' }, 'Unanimous')
|
|
1365
|
+
)
|
|
1366
|
+
)
|
|
1367
|
+
),
|
|
1368
|
+
|
|
1369
|
+
/* Composite Hybrid Synthesis (AST / Block Merge) */
|
|
1370
|
+
React.createElement(
|
|
1371
|
+
'div',
|
|
1372
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1373
|
+
React.createElement(
|
|
1374
|
+
'label',
|
|
1375
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1376
|
+
React.createElement('input', {
|
|
1377
|
+
type: 'checkbox',
|
|
1378
|
+
checked: Boolean(currentPreset.composite_merge_enabled),
|
|
1379
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, composite_merge_enabled: e.target.checked })),
|
|
1380
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1381
|
+
}),
|
|
1382
|
+
'Composite Hybrid Synthesis (AST / Block Merge)'
|
|
1383
|
+
),
|
|
1384
|
+
React.createElement(
|
|
1385
|
+
'div',
|
|
1386
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1387
|
+
'Synthesize modular code blocks, types, and algorithms from multiple candidates into a unified composite.'
|
|
1388
|
+
)
|
|
1389
|
+
),
|
|
1390
|
+
|
|
1391
|
+
/* Cost Budget Guardrails & Auto-Fallback */
|
|
1392
|
+
React.createElement(
|
|
1393
|
+
'div',
|
|
1394
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1395
|
+
React.createElement(
|
|
1396
|
+
'label',
|
|
1397
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1398
|
+
React.createElement('input', {
|
|
1399
|
+
type: 'checkbox',
|
|
1400
|
+
checked: Boolean(currentPreset.budget_guard_enabled),
|
|
1401
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, budget_guard_enabled: e.target.checked })),
|
|
1402
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1403
|
+
}),
|
|
1404
|
+
'Cost Budget Guardrails & Auto-Fallback'
|
|
1405
|
+
),
|
|
1406
|
+
React.createElement(
|
|
1407
|
+
'div',
|
|
1408
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1409
|
+
'Cap maximum run cost; auto-trim candidate pool or abort to prevent accidental token spend.'
|
|
1410
|
+
),
|
|
1411
|
+
currentPreset.budget_guard_enabled && React.createElement(
|
|
1412
|
+
'div',
|
|
1413
|
+
{ style: { marginLeft: 22, display: 'flex', alignItems: 'center', gap: 12, marginTop: 4 } },
|
|
1414
|
+
React.createElement(
|
|
1415
|
+
'div',
|
|
1416
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 6 } },
|
|
1417
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Max USD ($):'),
|
|
1418
|
+
React.createElement('input', {
|
|
1419
|
+
type: 'number',
|
|
1420
|
+
step: '0.01',
|
|
1421
|
+
className: 'moa-input',
|
|
1422
|
+
style: { width: 80, height: 28, padding: '0 6px', fontSize: 12 },
|
|
1423
|
+
value: currentPreset.max_budget_usd ?? 0.1,
|
|
1424
|
+
onChange: (e) => {
|
|
1425
|
+
const val = parseFloat(e.target.value)
|
|
1426
|
+
updateCurrentPreset((p) => ({ ...p, max_budget_usd: isNaN(val) ? 0 : val }))
|
|
1427
|
+
},
|
|
1428
|
+
})
|
|
1429
|
+
),
|
|
1430
|
+
React.createElement(
|
|
1431
|
+
'div',
|
|
1432
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 6 } },
|
|
1433
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, 'Action:'),
|
|
1434
|
+
React.createElement(
|
|
1435
|
+
'select',
|
|
1436
|
+
{
|
|
1437
|
+
className: 'moa-select',
|
|
1438
|
+
style: { width: 100, height: 28, fontSize: 12, padding: '0 6px' },
|
|
1439
|
+
value: currentPreset.budget_action || 'trim',
|
|
1440
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, budget_action: e.target.value })),
|
|
1441
|
+
},
|
|
1442
|
+
React.createElement('option', { value: 'trim' }, 'Trim Pool'),
|
|
1443
|
+
React.createElement('option', { value: 'abort' }, 'Abort')
|
|
1444
|
+
)
|
|
1445
|
+
)
|
|
1446
|
+
)
|
|
1447
|
+
),
|
|
1448
|
+
|
|
1449
|
+
/* Automated Benchmark & Post-Mortem Report */
|
|
1450
|
+
React.createElement(
|
|
1451
|
+
'div',
|
|
1452
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1453
|
+
React.createElement(
|
|
1454
|
+
'label',
|
|
1455
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1456
|
+
React.createElement('input', {
|
|
1457
|
+
type: 'checkbox',
|
|
1458
|
+
checked: Boolean(currentPreset.report_generation_enabled),
|
|
1459
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, report_generation_enabled: e.target.checked })),
|
|
1460
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1461
|
+
}),
|
|
1462
|
+
'Automated Benchmark & Post-Mortem Report'
|
|
1463
|
+
),
|
|
1464
|
+
React.createElement(
|
|
1465
|
+
'div',
|
|
1466
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1467
|
+
'Generate downloadable Markdown & JSON reports detailing candidate latency, agreement, test gate, and cost breakdown.'
|
|
1468
|
+
)
|
|
1469
|
+
),
|
|
1470
|
+
|
|
1471
|
+
/* Graceful Degradation & Local Fallback */
|
|
1472
|
+
React.createElement(
|
|
1473
|
+
'div',
|
|
1474
|
+
{ style: { display: 'flex', flexDirection: 'column', gap: 6, padding: '8px 12px', background: 'var(--dsw-alias-bg-layer-2)', borderRadius: 8, border: '1px solid var(--dsw-alias-border-l2)', marginTop: 8 } },
|
|
1475
|
+
React.createElement(
|
|
1476
|
+
'label',
|
|
1477
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500 } },
|
|
1478
|
+
React.createElement('input', {
|
|
1479
|
+
type: 'checkbox',
|
|
1480
|
+
checked: Boolean(currentPreset.local_fallback_enabled),
|
|
1481
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, local_fallback_enabled: e.target.checked })),
|
|
1482
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1483
|
+
}),
|
|
1484
|
+
'Graceful Degradation & Local Fallback'
|
|
1485
|
+
),
|
|
1486
|
+
React.createElement(
|
|
1487
|
+
'div',
|
|
1488
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
1489
|
+
'Automatically fall back to local Ollama / MiniPC models when online candidate APIs fail.'
|
|
1490
|
+
)
|
|
1491
|
+
),
|
|
1492
|
+
|
|
1330
1493
|
/* Fallback Judges Chain */
|
|
1331
1494
|
React.createElement(
|
|
1332
1495
|
'div',
|
|
@@ -1481,6 +1644,17 @@ window.__ModuleLoader__.load({
|
|
|
1481
1644
|
updateCurrentPreset((p) => ({ ...p, reference_timeout_sec: isNaN(val) ? 60 : val }))
|
|
1482
1645
|
},
|
|
1483
1646
|
})
|
|
1647
|
+
),
|
|
1648
|
+
React.createElement(
|
|
1649
|
+
'label',
|
|
1650
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 4 } },
|
|
1651
|
+
React.createElement('input', {
|
|
1652
|
+
type: 'checkbox',
|
|
1653
|
+
checked: Boolean(currentPreset.temperature_gradient_enabled),
|
|
1654
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, temperature_gradient_enabled: e.target.checked })),
|
|
1655
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
1656
|
+
}),
|
|
1657
|
+
'Temperature Gradient Exploration (0.2 → 0.9 across candidates)'
|
|
1484
1658
|
)
|
|
1485
1659
|
),
|
|
1486
1660
|
React.createElement(
|
|
@@ -1502,6 +1676,26 @@ window.__ModuleLoader__.load({
|
|
|
1502
1676
|
t,
|
|
1503
1677
|
})
|
|
1504
1678
|
),
|
|
1679
|
+
React.createElement('input', {
|
|
1680
|
+
type: 'number',
|
|
1681
|
+
step: '0.1',
|
|
1682
|
+
min: '0',
|
|
1683
|
+
max: '2',
|
|
1684
|
+
placeholder: 'Temp',
|
|
1685
|
+
title: 'Candidate Temperature (leave empty for default/gradient)',
|
|
1686
|
+
className: 'moa-input',
|
|
1687
|
+
style: { width: 50, height: 32, fontSize: 12, padding: '0 4px', textAlign: 'center' },
|
|
1688
|
+
value: (typeof ref.temperature === 'number' && ref.temperature >= 0) ? ref.temperature : '',
|
|
1689
|
+
onChange: (e) => {
|
|
1690
|
+
const raw = e.target.value
|
|
1691
|
+
const parsedVal = raw === '' ? -1 : parseFloat(raw)
|
|
1692
|
+
updateCurrentPreset((p) => {
|
|
1693
|
+
const refs = [...(p.reference_models || [])]
|
|
1694
|
+
refs[idx] = { ...refs[idx], temperature: isNaN(parsedVal) ? -1 : parsedVal }
|
|
1695
|
+
return { ...p, reference_models: refs }
|
|
1696
|
+
})
|
|
1697
|
+
},
|
|
1698
|
+
}),
|
|
1505
1699
|
React.createElement(
|
|
1506
1700
|
'select',
|
|
1507
1701
|
{
|
package/lib/history.js
CHANGED
|
@@ -86,6 +86,8 @@ function normalizeRunRecord(record) {
|
|
|
86
86
|
totalTokens: Number(record.totalTokens || 0),
|
|
87
87
|
totalCostUsd: Number(record.totalCostUsd || 0),
|
|
88
88
|
durationMs: Number(record.durationMs || 0),
|
|
89
|
+
benchmarkReport: record.benchmarkReport || null,
|
|
90
|
+
consensus: record.consensus || null,
|
|
89
91
|
}
|
|
90
92
|
}
|
|
91
93
|
|
|
@@ -413,6 +415,9 @@ export function candidatesForHistory(referenceOutputs) {
|
|
|
413
415
|
provider: r.slot?.provider || "",
|
|
414
416
|
model: r.slot?.model || "",
|
|
415
417
|
role: r.role_persona || r.slot?.role_persona || r.slot?.role || "general",
|
|
418
|
+
temperature: r.temperature,
|
|
419
|
+
wasFallback: Boolean(r.was_fallback),
|
|
420
|
+
originalModel: r.original_model || null,
|
|
416
421
|
files: (r.files || []).map((f) => f.relativePath),
|
|
417
422
|
usage: r.usage || { inputTokens: 0, outputTokens: 0 },
|
|
418
423
|
costUsd: r.costUsd || 0,
|
package/lib/index.js
CHANGED
|
@@ -32,6 +32,7 @@ import {
|
|
|
32
32
|
} from './moa-runner.js'
|
|
33
33
|
|
|
34
34
|
import { registerMoaRoutes } from './routes.js'
|
|
35
|
+
import { resolvePresetForPrompt } from './moa-router.js'
|
|
35
36
|
|
|
36
37
|
import { createLiveCanvasClient } from './live-canvas.js'
|
|
37
38
|
import { registerPluginUpdater, isSafeWriteRequest } from './updater.js'
|
|
@@ -52,6 +53,7 @@ export const ModelSlotSchema = z.object({
|
|
|
52
53
|
provider: z.string().default('opencode-go'),
|
|
53
54
|
model: z.string().default('deepseek-v4-flash'),
|
|
54
55
|
role_persona: z.string().default(''),
|
|
56
|
+
temperature: z.number().default(-1),
|
|
55
57
|
})
|
|
56
58
|
|
|
57
59
|
export const PresetSchema = z.object({
|
|
@@ -77,11 +79,27 @@ export const PresetSchema = z.object({
|
|
|
77
79
|
test_gate_enabled: z.boolean().default(false),
|
|
78
80
|
test_command: z.string().default(''),
|
|
79
81
|
test_gate_timeout_sec: z.number().default(15),
|
|
82
|
+
multi_judge_enabled: z.boolean().default(false),
|
|
83
|
+
judge_models: z.array(ModelSlotSchema).default([]),
|
|
84
|
+
judge_voting_strategy: z.string().default('majority'),
|
|
85
|
+
composite_merge_enabled: z.boolean().default(false),
|
|
86
|
+
smart_routing_enabled: z.boolean().default(false),
|
|
87
|
+
smart_routing_model: ModelSlotSchema.default({ provider: 'opencode-go', model: 'jev' }),
|
|
88
|
+
budget_guard_enabled: z.boolean().default(false),
|
|
89
|
+
max_budget_usd: z.number().default(0),
|
|
90
|
+
budget_action: z.string().default('trim'),
|
|
91
|
+
temperature_gradient_enabled: z.boolean().default(false),
|
|
92
|
+
multi_turn_enabled: z.boolean().default(true),
|
|
93
|
+
report_generation_enabled: z.boolean().default(false),
|
|
94
|
+
local_fallback_enabled: z.boolean().default(false),
|
|
95
|
+
local_fallback_models: z.array(ModelSlotSchema).default([]),
|
|
80
96
|
})
|
|
81
97
|
|
|
82
98
|
export const Config = z.object({
|
|
83
99
|
enabled: z.boolean().default(true).volatile(),
|
|
84
100
|
default_preset: z.string().default('default').volatile(),
|
|
101
|
+
smart_routing_enabled: z.boolean().default(false).volatile(),
|
|
102
|
+
smart_routing_model: ModelSlotSchema.default({ provider: 'opencode-go', model: 'jev' }).volatile(),
|
|
85
103
|
prices: z.dict(PriceRow).default({}).volatile(),
|
|
86
104
|
presets: z.array(PresetSchema).default([
|
|
87
105
|
{
|
|
@@ -338,6 +356,24 @@ export function apply(ctx, config) {
|
|
|
338
356
|
const cfg = live()
|
|
339
357
|
if (cfg.enabled !== false) {
|
|
340
358
|
const parsed = parseMoACommand(userText, cfg.presets || []) || { prompt: userText, presetName: cfg.default_preset || 'default' }
|
|
359
|
+
const isExplicitPreset = userText.includes('--preset') || userText.includes('-p ')
|
|
360
|
+
if (!isExplicitPreset && (cfg.smart_routing_enabled || cfg.presets?.some((p) => p.smart_routing_enabled))) {
|
|
361
|
+
try {
|
|
362
|
+
const routed = await resolvePresetForPrompt({
|
|
363
|
+
prompt: parsed.prompt || userText,
|
|
364
|
+
presets: cfg.presets || [],
|
|
365
|
+
defaultPreset: parsed.presetName || cfg.default_preset || 'default',
|
|
366
|
+
routingModel: cfg.smart_routing_model,
|
|
367
|
+
callLlm,
|
|
368
|
+
enabled: true,
|
|
369
|
+
})
|
|
370
|
+
if (routed?.isAutoRouted && routed.presetName) {
|
|
371
|
+
parsed.presetName = routed.presetName
|
|
372
|
+
}
|
|
373
|
+
} catch {
|
|
374
|
+
// fallback silently
|
|
375
|
+
}
|
|
376
|
+
}
|
|
341
377
|
const targetPreset = (cfg.presets || []).find((p) => p.name === parsed.presetName) || cfg.presets?.[0]
|
|
342
378
|
const aggProvider = targetPreset?.aggregator?.provider || 'codex'
|
|
343
379
|
const aggModel = targetPreset?.aggregator?.model || 'gpt-5.6-sol'
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cost Budget Guardrails & Auto-Trimming
|
|
3
|
+
* Feature 6 (Cost Budget Guardrails)
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import { estimateTokenCost } from './pricing.js'
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Calculates conservative upfront cost estimate for a list of candidate slots.
|
|
10
|
+
*/
|
|
11
|
+
export function estimateCandidateRunCost(slots = [], prices = {}, promptTokens = 500, expectedOutputTokens = 2048) {
|
|
12
|
+
let totalUsd = 0
|
|
13
|
+
const slotEstimates = []
|
|
14
|
+
|
|
15
|
+
for (const slot of slots) {
|
|
16
|
+
const costInfo = estimateTokenCost(
|
|
17
|
+
slot,
|
|
18
|
+
{ inputTokens: promptTokens, outputTokens: expectedOutputTokens },
|
|
19
|
+
prices
|
|
20
|
+
)
|
|
21
|
+
const cost = costInfo?.costUsd || 0
|
|
22
|
+
totalUsd += cost
|
|
23
|
+
slotEstimates.push({ slot, costUsd: cost })
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
return {
|
|
27
|
+
totalEstimatedUsd: Number(totalUsd.toFixed(6)),
|
|
28
|
+
slotEstimates,
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Applies budget guardrails before initiating parallel candidate execution.
|
|
34
|
+
*/
|
|
35
|
+
export function applyBudgetGuardrails({
|
|
36
|
+
references = [],
|
|
37
|
+
enabled = false,
|
|
38
|
+
maxBudgetUsd = 0,
|
|
39
|
+
action = 'trim', // 'trim' | 'abort'
|
|
40
|
+
prices = {},
|
|
41
|
+
promptLength = 1000,
|
|
42
|
+
}) {
|
|
43
|
+
if (!enabled || typeof maxBudgetUsd !== 'number' || maxBudgetUsd <= 0 || references.length === 0) {
|
|
44
|
+
return {
|
|
45
|
+
allowed: true,
|
|
46
|
+
references,
|
|
47
|
+
action: 'none',
|
|
48
|
+
estimatedCostUsd: 0,
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const promptTokens = Math.max(50, Math.round(promptLength / 4))
|
|
53
|
+
const { totalEstimatedUsd, slotEstimates } = estimateCandidateRunCost(references, prices, promptTokens)
|
|
54
|
+
|
|
55
|
+
if (totalEstimatedUsd <= maxBudgetUsd) {
|
|
56
|
+
return {
|
|
57
|
+
allowed: true,
|
|
58
|
+
references,
|
|
59
|
+
action: 'pass',
|
|
60
|
+
estimatedCostUsd: totalEstimatedUsd,
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
if (action === 'abort') {
|
|
65
|
+
return {
|
|
66
|
+
allowed: false,
|
|
67
|
+
references: [],
|
|
68
|
+
action: 'abort',
|
|
69
|
+
estimatedCostUsd: totalEstimatedUsd,
|
|
70
|
+
maxBudgetUsd,
|
|
71
|
+
reason: `Estimated MoA run cost ($${totalEstimatedUsd.toFixed(4)}) exceeds configured max budget ($${maxBudgetUsd.toFixed(4)}).`,
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Action is 'trim': Sort by estimated cost ascending (cheapest first)
|
|
76
|
+
const sorted = [...slotEstimates].sort((a, b) => a.costUsd - b.costUsd)
|
|
77
|
+
const kept = []
|
|
78
|
+
let cumulative = 0
|
|
79
|
+
|
|
80
|
+
for (const item of sorted) {
|
|
81
|
+
if (cumulative + item.costUsd <= maxBudgetUsd || kept.length === 0) {
|
|
82
|
+
kept.push(item.slot)
|
|
83
|
+
cumulative += item.costUsd
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
allowed: true,
|
|
89
|
+
references: kept,
|
|
90
|
+
action: 'trim',
|
|
91
|
+
originalCount: references.length,
|
|
92
|
+
trimmedCount: kept.length,
|
|
93
|
+
estimatedCostUsd: Number(cumulative.toFixed(6)),
|
|
94
|
+
maxBudgetUsd,
|
|
95
|
+
reason: `Trimmed candidate pool from ${references.length} to ${kept.length} models to stay within budget ($${maxBudgetUsd.toFixed(4)}).`,
|
|
96
|
+
}
|
|
97
|
+
}
|
package/lib/moa-candidates.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* DeepSeek Harness Mixture of Agents (MoA) — Candidate Fan-Out & Retry
|
|
3
|
+
* With Per-Candidate Temperature, Temperature Gradient & Local Fallback Resilience
|
|
3
4
|
*/
|
|
4
5
|
|
|
5
6
|
import { bestEffort } from './best-effort.js'
|
|
@@ -29,7 +30,8 @@ export async function callWithTransientRetry(callLlmFn, callArgs, maxRetries = 0
|
|
|
29
30
|
}
|
|
30
31
|
|
|
31
32
|
/**
|
|
32
|
-
* Dispatches queries to all reference models in parallel with transient retry
|
|
33
|
+
* Dispatches queries to all reference models in parallel with transient retry,
|
|
34
|
+
* per-candidate temperature / gradient exploration, local fallback, and quorum mitigation.
|
|
33
35
|
*/
|
|
34
36
|
export async function runReferencesParallel(references, messages, options = {}, callLlm, onProgress) {
|
|
35
37
|
if (!Array.isArray(references) || references.length === 0) {
|
|
@@ -87,6 +89,14 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
87
89
|
: ''
|
|
88
90
|
const slotMessages = [{ role: 'system', content: systemPrompt + persona }, ...advisoryMessages]
|
|
89
91
|
|
|
92
|
+
// Feature 7: Per-candidate temperature or temperature gradient
|
|
93
|
+
let effectiveTemperature = options.temperature ?? 0.6
|
|
94
|
+
if (typeof slot.temperature === 'number' && slot.temperature >= 0) {
|
|
95
|
+
effectiveTemperature = slot.temperature
|
|
96
|
+
} else if (options.temperature_gradient_enabled && total > 1) {
|
|
97
|
+
effectiveTemperature = Number((0.2 + (0.7 * i) / (total - 1)).toFixed(2))
|
|
98
|
+
}
|
|
99
|
+
|
|
90
100
|
const runOne = async () => {
|
|
91
101
|
try {
|
|
92
102
|
const callPromise = callWithTransientRetry(
|
|
@@ -95,7 +105,7 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
95
105
|
provider: slot.provider,
|
|
96
106
|
model: slot.model,
|
|
97
107
|
messages: slotMessages,
|
|
98
|
-
temperature:
|
|
108
|
+
temperature: effectiveTemperature,
|
|
99
109
|
maxTokens: options.maxTokens ?? 4096,
|
|
100
110
|
timeoutMs,
|
|
101
111
|
signal: abortControllers[i].signal,
|
|
@@ -124,38 +134,96 @@ export async function runReferencesParallel(references, messages, options = {},
|
|
|
124
134
|
const costInfo = estimateTokenCost(slot, fallbackUsage, options.prices)
|
|
125
135
|
|
|
126
136
|
finishedCount++
|
|
127
|
-
if (typeof onProgress === 'function') {
|
|
128
|
-
const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
|
|
129
|
-
onProgress(`✅ *Candidate ${i + 1}/${total} (${label}) finished${costStr}*\n`)
|
|
130
|
-
}
|
|
131
|
-
|
|
132
137
|
results[i] = {
|
|
133
138
|
index: i + 1,
|
|
134
139
|
slot,
|
|
135
140
|
label,
|
|
136
141
|
role_persona: role,
|
|
142
|
+
temperature: effectiveTemperature,
|
|
137
143
|
text,
|
|
138
144
|
usage: costInfo,
|
|
139
145
|
costUsd: costInfo.costUsd,
|
|
140
146
|
ok: true,
|
|
141
147
|
}
|
|
142
|
-
|
|
143
|
-
finishedCount++
|
|
144
|
-
const errMsg = err?.message || String(err)
|
|
145
|
-
console.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
|
|
148
|
+
|
|
146
149
|
if (typeof onProgress === 'function') {
|
|
147
|
-
|
|
150
|
+
const costStr = costInfo.costUsd > 0 ? ` (~\$${costInfo.costUsd.toFixed(4)})` : ''
|
|
151
|
+
const currentTotalCost = results.reduce((sum, r) => sum + (r?.costUsd || 0), 0)
|
|
152
|
+
onProgress(`✅ *Candidate ${i + 1}/${total} (${label}, T=${effectiveTemperature}) finished${costStr}* | Live total: ~\$${currentTotalCost.toFixed(4)}\n`)
|
|
148
153
|
}
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
154
|
+
} catch (err) {
|
|
155
|
+
// Feature 10: Local fallback models if online candidate fails
|
|
156
|
+
let fallbackRecovered = false
|
|
157
|
+
if (options.local_fallback_enabled && Array.isArray(options.local_fallback_models) && options.local_fallback_models.length > 0) {
|
|
158
|
+
for (const fbSlot of options.local_fallback_models) {
|
|
159
|
+
const fbLabel = slotLabel(fbSlot)
|
|
160
|
+
try {
|
|
161
|
+
if (typeof onProgress === 'function') {
|
|
162
|
+
onProgress(`🔄 *Candidate ${i + 1} (${label}) failed, attempting local fallback to ${fbLabel}...*\n`)
|
|
163
|
+
}
|
|
164
|
+
const fbRes = await callWithTransientRetry(
|
|
165
|
+
callLlm,
|
|
166
|
+
{
|
|
167
|
+
provider: fbSlot.provider,
|
|
168
|
+
model: fbSlot.model,
|
|
169
|
+
messages: slotMessages,
|
|
170
|
+
temperature: effectiveTemperature,
|
|
171
|
+
maxTokens: options.maxTokens ?? 4096,
|
|
172
|
+
timeoutMs,
|
|
173
|
+
signal: abortControllers[i].signal,
|
|
174
|
+
},
|
|
175
|
+
0
|
|
176
|
+
)
|
|
177
|
+
const fbText = typeof fbRes === 'string' ? fbRes : (fbRes?.content || fbRes?.text || '')
|
|
178
|
+
const fbUsage = (typeof fbRes === 'object' && fbRes?.usage) ? fbRes.usage : {
|
|
179
|
+
inputTokens: Math.max(1, Math.round(slotMessages.map((m) => m.content).join('').length / 4)),
|
|
180
|
+
outputTokens: Math.max(1, Math.round(fbText.length / 4)),
|
|
181
|
+
}
|
|
182
|
+
const fbCost = estimateTokenCost(fbSlot, fbUsage, options.prices)
|
|
183
|
+
finishedCount++
|
|
184
|
+
results[i] = {
|
|
185
|
+
index: i + 1,
|
|
186
|
+
slot: fbSlot,
|
|
187
|
+
label: `${fbLabel} [fallback for ${label}]`,
|
|
188
|
+
role_persona: role,
|
|
189
|
+
temperature: effectiveTemperature,
|
|
190
|
+
text: fbText,
|
|
191
|
+
usage: fbCost,
|
|
192
|
+
costUsd: fbCost.costUsd,
|
|
193
|
+
ok: true,
|
|
194
|
+
was_fallback: true,
|
|
195
|
+
original_model: label,
|
|
196
|
+
}
|
|
197
|
+
fallbackRecovered = true
|
|
198
|
+
if (typeof onProgress === 'function') {
|
|
199
|
+
onProgress(`✅ *Candidate ${i + 1}/${total} recovered using fallback (${fbLabel})*\n`)
|
|
200
|
+
}
|
|
201
|
+
break
|
|
202
|
+
} catch (fbErr) {
|
|
203
|
+
console.warn(`[dsh-moa] Fallback ${fbLabel} also failed:`, fbErr?.message || fbErr)
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (!fallbackRecovered) {
|
|
209
|
+
finishedCount++
|
|
210
|
+
const errMsg = err?.message || String(err)
|
|
211
|
+
console.warn(`[dsh-moa] Reference ${label} failed:`, errMsg)
|
|
212
|
+
if (typeof onProgress === 'function') {
|
|
213
|
+
onProgress(`⚠️ *Candidate ${i + 1}/${total} (${label}) error: ${errMsg}*\n`)
|
|
214
|
+
}
|
|
215
|
+
results[i] = {
|
|
216
|
+
index: i + 1,
|
|
217
|
+
slot,
|
|
218
|
+
label,
|
|
219
|
+
role_persona: role,
|
|
220
|
+
temperature: effectiveTemperature,
|
|
221
|
+
text: `[Model ${label} error: ${errMsg}]`,
|
|
222
|
+
usage: { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: 0 },
|
|
223
|
+
costUsd: 0,
|
|
224
|
+
ok: false,
|
|
225
|
+
error: errMsg,
|
|
226
|
+
}
|
|
159
227
|
}
|
|
160
228
|
} finally {
|
|
161
229
|
notifyFinished()
|