@goodandready/dsh-moa 0.2.10 → 0.2.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/docs/README.ru.md +3 -0
- package/docs/design/DESIGN.md +3 -0
- package/docs/plans/55-resilience-plan.md +33 -0
- package/lib/client.js +116 -0
- package/lib/history.js +87 -35
- package/lib/index.js +6 -2
- package/lib/moa-parser.js +181 -0
- package/lib/moa-prompts.js +270 -0
- package/lib/moa-runner.js +150 -493
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -199,6 +199,9 @@ dsh-moa:
|
|
|
199
199
|
| `presets[].quorum_enabled` | `boolean` | `false` | Straggler mitigation: proceed with synthesis once >= 60% candidates respond |
|
|
200
200
|
| `presets[].grace_period_sec` | `number` | `10` | Grace period in seconds to wait for stragglers after quorum is reached |
|
|
201
201
|
| `presets[].aggregator_fallbacks` | `array` | `[]` | Ordered fallback judge models tried if primary aggregator encounters transient errors |
|
|
202
|
+
| `presets[].blind_evaluation` | `boolean` | `false` | Anonymize candidate model names for the judge/curator to eliminate family/brand bias |
|
|
203
|
+
| `presets[].reference_timeout_sec` | `number` | `60` | Per-candidate execution timeout in seconds |
|
|
204
|
+
| `presets[].aggregator_timeout_sec` | `number` | `180` | Aggregator/judge synthesis timeout in seconds |
|
|
202
205
|
| `presets[].reference_temperature` / `.aggregator_temperature` | `number` | `0.6` / `0.4` | Sampling temperatures for proposers and judge |
|
|
203
206
|
| `presets[].max_tokens` | `number` | `4096` | Max output tokens per model call |
|
|
204
207
|
| `presets[].judge_criteria` | `string` | `""` | Optional extra evaluation criteria passed to the judge |
|
package/docs/README.ru.md
CHANGED
|
@@ -199,6 +199,9 @@ dsh-moa:
|
|
|
199
199
|
| `presets[].quorum_enabled` | `boolean` | `false` | Защита от зависших моделей (stragglers): запуск синтеза при ответе от >= 60% кандидатов |
|
|
200
200
|
| `presets[].grace_period_sec` | `number` | `10` | Грейс-период (в секундах) ожидания оставшихся моделей после достижения кворума |
|
|
201
201
|
| `presets[].aggregator_fallbacks` | `array` | `[]` | Список запасных моделей-судей при сбоях основной модели агрегатора |
|
|
202
|
+
| `presets[].blind_evaluation` | `boolean` | `false` | Обезличивание имен кандидатов («Candidate 1», «Candidate 2») для исключения предвзятости судьи |
|
|
203
|
+
| `presets[].reference_timeout_sec` | `number` | `60` | Таймаут опроса каждого кандидата в секундах |
|
|
204
|
+
| `presets[].aggregator_timeout_sec` | `number` | `180` | Таймаут синтеза решения судьей в секундах |
|
|
202
205
|
| `presets[].reference_temperature` / `.aggregator_temperature` | `number` | `0.6` / `0.4` | Температуры сэмплирования советников и судьи |
|
|
203
206
|
| `presets[].max_tokens` | `number` | `4096` | Максимум выходных токенов на вызов модели |
|
|
204
207
|
| `presets[].judge_criteria` | `string` | `""` | Опциональные дополнительные критерии оценки для судьи |
|
package/docs/design/DESIGN.md
CHANGED
|
@@ -58,9 +58,12 @@
|
|
|
58
58
|
- 2026-09-10 — Унификация UI с дизайн-стандартом `dsh-clinebot` (статусные чипы в header, секционные карточки, дизайн-токены `--dsw-alias-*`, глубокий аудит устойчивости).
|
|
59
59
|
- 2026-09-11 — English-canonical пользовательские строки (сервер и клиент); ru-перевод предоставляет translation-плагин, собственный ru-дубль из пакета удалён. Причина: стандарт DSH (dsh-plugin-authoring); пересмотр — только по явному решению владельца.
|
|
60
60
|
- 2026-09-11 — Заглушка Live Canvas удалена (фабрикация `http://localhost:3000/preview/...`); заявления README сняты. Реальная интеграция с `dsh-live-canvas` — отдельная задача (Gitea #46). Changed: прежний пункт про автопревью больше не действует.
|
|
61
|
+
- 2026-09-12 — Реализован режим слепого судейства (`blind_evaluation: boolean`), гарантированное зеркалирование языка запроса (Language Mirroring Guard в промптах судьи и куратора), интерактивный лидерборд моделей (винрейт, запуски, победы, средняя стоимость) в секции аналитики UI и раздельные настройки таймаутов кандидатов и судьи в карточке настроек (Gitea Issue #57).
|
|
61
62
|
- 2026-09-11 — Статусный бейдж карточки отражает фактический `GET /dsh-moa/status` (online / offline / disabled); телеметрия Total Runs / Avg Run Cost берётся из `GET /dsh-moa/history`. Причина: карточка не должна показывать состояния, которые она не проверяла.
|
|
62
63
|
- 2026-09-11 — Переключатель `enabled` в карточке сохраняется через `settingsScope`/REST и влияет на `/moa`-turn и `POST /dsh-moa/run`.
|
|
63
64
|
- 2026-09-11 (вечер) — Интеграция с Live Canvas реализована через собственный REST-контракт `@goodandready/dsh-live-canvas` (`POST /dsh-live-canvas/api/preview`, self-call на порт хоста `ctx.webServer.port`) с тихой деградацией при отсутствии плагина (любая ошибка → ответ без preview-ссылки). Заменяет прежнее решение об удалении фабрикованной заглушки: ссылка `/dsh-live-canvas/sandbox/<id>` теперь создаётся реальной песочницей.
|
|
64
65
|
|
|
65
66
|
|
|
66
67
|
- 2026-09-12 — Реализация кураторского синтеза (`curator_synthesis`), строгой рубрики антипаттернов (`ANTIPATTERNS_RUBRIC`), живого потокового стриминга куратора (`stream_aggregator`), кворума кандидатов с льготным периодом (`quorum_enabled`, `grace_period_sec`), авторетрая транзиентных ошибок (`callWithTransientRetry`) и цепочки запасных судей (`aggregator_fallbacks`). Все опции конфигурируются в пресетах с сохранением 100% обратной совместимости.
|
|
68
|
+
|
|
69
|
+
- 2026-09-12 (вечер) — Декомпозиция lib/moa-runner.js на специализированные модули (`lib/moa-prompts.js`, `lib/moa-parser.js`, `lib/moa-runner.js`) с соблюдением канонического лимита <= 800 строк. Реализация автоматической ротации истории при превышении 10 МБ (`rotateHistoryFileIfNeeded`) и неблокирующего асинхронного сохранения (`recordMoaRunAsync`). Настраиваемые таймауты кандидатов и агрегатора (`reference_timeout_sec`, `aggregator_timeout_sec`).
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Task Plan: MoA Resilience, Async I/O, Configurable Timeouts and Modularization
|
|
2
|
+
|
|
3
|
+
Issue: #55
|
|
4
|
+
Branch: `feat/moa-resilience-refactor`
|
|
5
|
+
Worktree: `/mnt/external/Project/DEV/dhsplugins/dsh-moa/.worktrees/feat/moa-resilience-refactor`
|
|
6
|
+
|
|
7
|
+
## Objectives
|
|
8
|
+
1. Modularize `lib/moa-runner.js` into clean, focused modules under 800 lines:
|
|
9
|
+
- `lib/moa-prompts.js`: Rubric, system prompts, synthesis and questionnaire prompts.
|
|
10
|
+
- `lib/moa-parser.js`: Output parsing, winner extraction, recommended assembler, file block extraction.
|
|
11
|
+
- `lib/moa-runner.js`: Focused pipeline orchestrator.
|
|
12
|
+
2. Optimize storage in `lib/history.js`:
|
|
13
|
+
- Non-blocking async append via `fs.promises.appendFile`.
|
|
14
|
+
- File size rotation (>10MB or >5000 lines -> `.jsonl.1` backup).
|
|
15
|
+
- Fast tail reader for paginated history to avoid loading gigantic files in memory.
|
|
16
|
+
3. Configurable timeouts in `lib/index.js` and `PresetSchema`:
|
|
17
|
+
- `reference_timeout_sec` (default: 60s).
|
|
18
|
+
- `aggregator_timeout_sec` (default: 180s).
|
|
19
|
+
- Forward timeouts into `callLlm` and `AbortController`.
|
|
20
|
+
4. Fallback recovery on malformed judge output:
|
|
21
|
+
- When judge output is empty or truncated, cleanly fall back to `aggregator_fallbacks` before degrading.
|
|
22
|
+
5. Verification & Tests:
|
|
23
|
+
- Add unit tests in `test/moa-resilience.test.mjs`.
|
|
24
|
+
- Verify existing 60/60 tests pass without regression.
|
|
25
|
+
- Update `docs/design/DESIGN.md` and README files.
|
|
26
|
+
|
|
27
|
+
## Phases
|
|
28
|
+
- [x] Phase 1: Planning & Setup (Issue #55, isolated worktree created, test suite verified)
|
|
29
|
+
- [ ] Phase 2: Modularization of moa-runner (`lib/moa-prompts.js`, `lib/moa-parser.js`)
|
|
30
|
+
- [ ] Phase 3: History Storage Rotation & Async I/O (`lib/history.js`)
|
|
31
|
+
- [ ] Phase 4: Configurable Timeouts & Fallback Recovery (`lib/index.js`, `lib/moa-runner.js`)
|
|
32
|
+
- [ ] Phase 5: Test Coverage & Verification (`test/moa-resilience.test.mjs`)
|
|
33
|
+
- [ ] Phase 6: Documentation, PR, Merge, Release & Deploy
|
package/lib/client.js
CHANGED
|
@@ -55,6 +55,18 @@ window.__ModuleLoader__.load({
|
|
|
55
55
|
'aggregator.curator_hint': 'When enabled: the curator highlights the finest components of each candidate, applies the antipatterns rubric, and advises which agent model should assemble the solution.',
|
|
56
56
|
'aggregator.stream_label': 'Live Stream Curator/Judge Thinking',
|
|
57
57
|
'aggregator.stream_hint': 'Streams aggregator tokens directly to chat in real time for zero-latency initial response.',
|
|
58
|
+
'aggregator.blind_label': 'Blind Review (Anonymize Candidates for Judge)',
|
|
59
|
+
'aggregator.blind_hint': 'When enabled: candidate model names and providers are hidden from the judge prompt to prevent family/brand bias.',
|
|
60
|
+
'aggregator.timeout_label': 'Judge synthesis timeout (sec):',
|
|
61
|
+
'proposers.timeout_label': 'Candidate timeout (sec):',
|
|
62
|
+
'leaderboard.title': '🏆 Model Win-Rate Leaderboard',
|
|
63
|
+
'leaderboard.desc': 'Performance metrics calculated from recorded MoA run verdicts.',
|
|
64
|
+
'leaderboard.model': 'Model',
|
|
65
|
+
'leaderboard.runs': 'Runs',
|
|
66
|
+
'leaderboard.wins': 'Wins',
|
|
67
|
+
'leaderboard.winrate': 'Win Rate',
|
|
68
|
+
'leaderboard.cost': 'Avg Cost',
|
|
69
|
+
'leaderboard.empty': 'No historical runs recorded yet.',
|
|
58
70
|
'aggregator.fallbacks_title': 'Fallback Judges Chain',
|
|
59
71
|
'aggregator.fallbacks_desc': 'Sequential backup models invoked automatically if the primary judge experiences rate limits or outages.',
|
|
60
72
|
'aggregator.add_fallback_btn': '+ Add Fallback Judge',
|
|
@@ -448,6 +460,7 @@ window.__ModuleLoader__.load({
|
|
|
448
460
|
setEnabled,
|
|
449
461
|
hostStatus,
|
|
450
462
|
stats,
|
|
463
|
+
leaderboard = [],
|
|
451
464
|
handleSave,
|
|
452
465
|
saveStatus,
|
|
453
466
|
reload,
|
|
@@ -831,6 +844,39 @@ window.__ModuleLoader__.load({
|
|
|
831
844
|
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
832
845
|
t('aggregator.stream_hint')
|
|
833
846
|
),
|
|
847
|
+
React.createElement(
|
|
848
|
+
'label',
|
|
849
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, cursor: 'pointer', fontSize: 13, fontWeight: 500, marginTop: 8 } },
|
|
850
|
+
React.createElement('input', {
|
|
851
|
+
type: 'checkbox',
|
|
852
|
+
checked: Boolean(currentPreset.blind_evaluation),
|
|
853
|
+
onChange: (e) => updateCurrentPreset((p) => ({ ...p, blind_evaluation: e.target.checked })),
|
|
854
|
+
style: { accentColor: 'var(--dsw-alias-state-brand-primary, var(--dsw-alias-label-primary))', cursor: 'pointer' },
|
|
855
|
+
}),
|
|
856
|
+
t('aggregator.blind_label')
|
|
857
|
+
),
|
|
858
|
+
React.createElement(
|
|
859
|
+
'div',
|
|
860
|
+
{ style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginLeft: 22 } },
|
|
861
|
+
t('aggregator.blind_hint')
|
|
862
|
+
),
|
|
863
|
+
React.createElement(
|
|
864
|
+
'div',
|
|
865
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, marginTop: 8 } },
|
|
866
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('aggregator.timeout_label')),
|
|
867
|
+
React.createElement('input', {
|
|
868
|
+
type: 'number',
|
|
869
|
+
className: 'moa-input',
|
|
870
|
+
style: { width: 80, height: 28, padding: '0 6px', fontSize: 12 },
|
|
871
|
+
min: 30,
|
|
872
|
+
max: 600,
|
|
873
|
+
value: currentPreset.aggregator_timeout_sec ?? 180,
|
|
874
|
+
onChange: (e) => {
|
|
875
|
+
const val = parseInt(e.target.value, 10)
|
|
876
|
+
updateCurrentPreset((p) => ({ ...p, aggregator_timeout_sec: isNaN(val) ? 180 : val }))
|
|
877
|
+
},
|
|
878
|
+
})
|
|
879
|
+
),
|
|
834
880
|
/* Fallback Judges Chain */
|
|
835
881
|
React.createElement(
|
|
836
882
|
'div',
|
|
@@ -968,6 +1014,23 @@ window.__ModuleLoader__.load({
|
|
|
968
1014
|
updateCurrentPreset((p) => ({ ...p, grace_period_sec: isNaN(val) ? 10 : val }))
|
|
969
1015
|
},
|
|
970
1016
|
})
|
|
1017
|
+
),
|
|
1018
|
+
React.createElement(
|
|
1019
|
+
'div',
|
|
1020
|
+
{ style: { display: 'flex', alignItems: 'center', gap: 8, marginTop: 4 } },
|
|
1021
|
+
React.createElement('span', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)' } }, t('proposers.timeout_label')),
|
|
1022
|
+
React.createElement('input', {
|
|
1023
|
+
type: 'number',
|
|
1024
|
+
className: 'moa-input',
|
|
1025
|
+
style: { width: 70, height: 28, padding: '0 6px', fontSize: 12 },
|
|
1026
|
+
min: 10,
|
|
1027
|
+
max: 300,
|
|
1028
|
+
value: currentPreset.reference_timeout_sec ?? 60,
|
|
1029
|
+
onChange: (e) => {
|
|
1030
|
+
const val = parseInt(e.target.value, 10)
|
|
1031
|
+
updateCurrentPreset((p) => ({ ...p, reference_timeout_sec: isNaN(val) ? 60 : val }))
|
|
1032
|
+
},
|
|
1033
|
+
})
|
|
971
1034
|
)
|
|
972
1035
|
),
|
|
973
1036
|
React.createElement(
|
|
@@ -1072,6 +1135,48 @@ window.__ModuleLoader__.load({
|
|
|
1072
1135
|
React.createElement('div', { className: 'moa-stat-val' }, String(currentRefs.length)),
|
|
1073
1136
|
React.createElement('div', { className: 'moa-stat-lbl' }, t('stats.active_models'))
|
|
1074
1137
|
)
|
|
1138
|
+
),
|
|
1139
|
+
Array.isArray(leaderboard) && leaderboard.length > 0 && React.createElement(
|
|
1140
|
+
'div',
|
|
1141
|
+
{ style: { marginTop: 12, display: 'flex', flexDirection: 'column', gap: 6 } },
|
|
1142
|
+
React.createElement('div', { style: { fontSize: 13, fontWeight: 600, color: 'var(--dsw-alias-label-primary)' } }, t('leaderboard.title')),
|
|
1143
|
+
React.createElement('div', { style: { fontSize: 12, color: 'var(--dsw-alias-label-secondary)', marginTop: -4 } }, t('leaderboard.desc')),
|
|
1144
|
+
React.createElement(
|
|
1145
|
+
'div',
|
|
1146
|
+
{ style: { overflowX: 'auto', border: '1px solid var(--dsw-alias-border-l2)', borderRadius: 8 } },
|
|
1147
|
+
React.createElement(
|
|
1148
|
+
'table',
|
|
1149
|
+
{ style: { width: '100%', borderCollapse: 'collapse', fontSize: 12, textAlign: 'left' } },
|
|
1150
|
+
React.createElement(
|
|
1151
|
+
'thead',
|
|
1152
|
+
{ style: { background: 'var(--dsw-alias-bg-layer-2)', borderBottom: '1px solid var(--dsw-alias-border-l2)' } },
|
|
1153
|
+
React.createElement(
|
|
1154
|
+
'tr',
|
|
1155
|
+
null,
|
|
1156
|
+
React.createElement('th', { style: { padding: '6px 10px' } }, t('leaderboard.model')),
|
|
1157
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.runs')),
|
|
1158
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.wins')),
|
|
1159
|
+
React.createElement('th', { style: { padding: '6px 8px' } }, t('leaderboard.winrate')),
|
|
1160
|
+
React.createElement('th', { style: { padding: '6px 10px' } }, t('leaderboard.cost'))
|
|
1161
|
+
)
|
|
1162
|
+
),
|
|
1163
|
+
React.createElement(
|
|
1164
|
+
'tbody',
|
|
1165
|
+
null,
|
|
1166
|
+
leaderboard.map((m, mIdx) =>
|
|
1167
|
+
React.createElement(
|
|
1168
|
+
'tr',
|
|
1169
|
+
{ key: mIdx, style: { borderBottom: mIdx < leaderboard.length - 1 ? '1px solid var(--dsw-alias-border-l2)' : 'none' } },
|
|
1170
|
+
React.createElement('td', { style: { padding: '6px 10px', fontWeight: 500, color: 'var(--dsw-alias-label-primary)' } }, m.modelKey || m.model),
|
|
1171
|
+
React.createElement('td', { style: { padding: '6px 8px', color: 'var(--dsw-alias-label-secondary)' } }, String(m.runs)),
|
|
1172
|
+
React.createElement('td', { style: { padding: '6px 8px', color: 'var(--dsw-alias-state-success-primary)' } }, String(m.wins)),
|
|
1173
|
+
React.createElement('td', { style: { padding: '6px 8px', fontWeight: 600 } }, (m.winRate ?? 0) + '%'),
|
|
1174
|
+
React.createElement('td', { style: { padding: '6px 10px', color: 'var(--dsw-alias-label-secondary)' } }, (m.avgCostUsd > 0) ? ('$' + m.avgCostUsd.toFixed(4)) : 'Free')
|
|
1175
|
+
)
|
|
1176
|
+
)
|
|
1177
|
+
)
|
|
1178
|
+
)
|
|
1179
|
+
)
|
|
1075
1180
|
)
|
|
1076
1181
|
),
|
|
1077
1182
|
|
|
@@ -1151,6 +1256,7 @@ window.__ModuleLoader__.load({
|
|
|
1151
1256
|
const [enabled, setEnabled] = React.useState(true)
|
|
1152
1257
|
const [hostStatus, setHostStatus] = React.useState('connecting')
|
|
1153
1258
|
const [stats, setStats] = React.useState({ totalRuns: null, avgCostUsd: null })
|
|
1259
|
+
const [leaderboard, setLeaderboard] = React.useState([])
|
|
1154
1260
|
|
|
1155
1261
|
React.useEffect(() => {
|
|
1156
1262
|
if (!scope || !snapshot) return
|
|
@@ -1182,6 +1288,15 @@ window.__ModuleLoader__.load({
|
|
|
1182
1288
|
setHostStatus('offline')
|
|
1183
1289
|
})
|
|
1184
1290
|
|
|
1291
|
+
fetch('/dsh-moa/leaderboard', { cache: 'no-store' })
|
|
1292
|
+
.then((r) => r.json())
|
|
1293
|
+
.then((data) => {
|
|
1294
|
+
if (data && data.ok && Array.isArray(data.models)) {
|
|
1295
|
+
setLeaderboard(data.models.slice(0, 10))
|
|
1296
|
+
}
|
|
1297
|
+
})
|
|
1298
|
+
.catch(() => {})
|
|
1299
|
+
|
|
1185
1300
|
fetch('/dsh-moa/history?limit=100', { cache: 'no-store' })
|
|
1186
1301
|
.then((r) => r.json())
|
|
1187
1302
|
.then((data) => {
|
|
@@ -1290,6 +1405,7 @@ window.__ModuleLoader__.load({
|
|
|
1290
1405
|
setEnabled,
|
|
1291
1406
|
hostStatus,
|
|
1292
1407
|
stats,
|
|
1408
|
+
leaderboard,
|
|
1293
1409
|
handleSave,
|
|
1294
1410
|
saveStatus,
|
|
1295
1411
|
reload,
|
package/lib/history.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* MoA Run History & Analytics Module
|
|
3
3
|
* Records every MoA pipeline execution, tracks model win rates and costs,
|
|
4
|
-
* and provides history search and
|
|
4
|
+
* and provides history search, file rotation, and non-blocking I/O.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import fs from 'node:fs'
|
|
@@ -12,6 +12,9 @@ import crypto from 'node:crypto'
|
|
|
12
12
|
const DEFAULT_HISTORY_DIR = path.join(os.homedir(), '.dsh')
|
|
13
13
|
const DEFAULT_HISTORY_FILE = path.join(DEFAULT_HISTORY_DIR, 'moa-history.jsonl')
|
|
14
14
|
|
|
15
|
+
/** Max history file size before rotation (10 MB) */
|
|
16
|
+
export const MAX_HISTORY_BYTES = 10 * 1024 * 1024
|
|
17
|
+
|
|
15
18
|
export function getHistoryFilePath(customDir) {
|
|
16
19
|
if (customDir) {
|
|
17
20
|
return path.join(customDir, '.moa-history.jsonl')
|
|
@@ -20,7 +23,67 @@ export function getHistoryFilePath(customDir) {
|
|
|
20
23
|
}
|
|
21
24
|
|
|
22
25
|
/**
|
|
23
|
-
*
|
|
26
|
+
* Checks if the history file exceeds the size threshold and rotates it.
|
|
27
|
+
* Moves current file to `${filePath}.1`, removing any older `.1` backup.
|
|
28
|
+
*/
|
|
29
|
+
export function rotateHistoryFileIfNeeded(filePath = DEFAULT_HISTORY_FILE, maxBytes = MAX_HISTORY_BYTES) {
|
|
30
|
+
try {
|
|
31
|
+
if (!fs.existsSync(filePath)) return false
|
|
32
|
+
const stat = fs.statSync(filePath)
|
|
33
|
+
if (stat.size >= maxBytes) {
|
|
34
|
+
const backupPath = `${filePath}.1`
|
|
35
|
+
if (fs.existsSync(backupPath)) {
|
|
36
|
+
try { fs.unlinkSync(backupPath) } catch {}
|
|
37
|
+
}
|
|
38
|
+
fs.renameSync(filePath, backupPath)
|
|
39
|
+
return true
|
|
40
|
+
}
|
|
41
|
+
} catch (err) {
|
|
42
|
+
console.warn('[dsh-moa] History rotation warning:', err?.message || err)
|
|
43
|
+
}
|
|
44
|
+
return false
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Prepares a normalized history record entry.
|
|
49
|
+
*/
|
|
50
|
+
function normalizeRunRecord(record) {
|
|
51
|
+
return {
|
|
52
|
+
id: record.id || crypto.randomUUID(),
|
|
53
|
+
timestamp: record.timestamp || new Date().toISOString(),
|
|
54
|
+
prompt: record.prompt || '',
|
|
55
|
+
preset: record.preset || 'default',
|
|
56
|
+
isRefinement: Boolean(record.isRefinement),
|
|
57
|
+
isFastMode: Boolean(record.isFastMode),
|
|
58
|
+
candidates: Array.isArray(record.candidates)
|
|
59
|
+
? record.candidates.map((c) => ({
|
|
60
|
+
provider: c.provider || '',
|
|
61
|
+
model: c.model || '',
|
|
62
|
+
filesCount: Array.isArray(c.files) ? c.files.length : (c.filesCount || 0),
|
|
63
|
+
usage: c.usage || { inputTokens: 0, outputTokens: 0 },
|
|
64
|
+
costUsd: Number(c.costUsd || 0),
|
|
65
|
+
}))
|
|
66
|
+
: [],
|
|
67
|
+
aggregator: record.aggregator
|
|
68
|
+
? {
|
|
69
|
+
provider: record.aggregator.provider || '',
|
|
70
|
+
model: record.aggregator.model || '',
|
|
71
|
+
usage: record.aggregator.usage || { inputTokens: 0, outputTokens: 0 },
|
|
72
|
+
costUsd: Number(record.aggregator.costUsd || 0),
|
|
73
|
+
}
|
|
74
|
+
: null,
|
|
75
|
+
winnerIndex: typeof record.winnerIndex === 'number' ? record.winnerIndex : -1,
|
|
76
|
+
winnerModel: record.winnerModel || '',
|
|
77
|
+
promotedFiles: Array.isArray(record.promotedFiles) ? record.promotedFiles : [],
|
|
78
|
+
totalTokens: Number(record.totalTokens || 0),
|
|
79
|
+
totalCostUsd: Number(record.totalCostUsd || 0),
|
|
80
|
+
durationMs: Number(record.durationMs || 0),
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Appends a completed MoA run record to the history file synchronously (preserves sync API).
|
|
86
|
+
* Performs automatic size-based rotation.
|
|
24
87
|
*/
|
|
25
88
|
export function recordMoaRun(record, filePath = DEFAULT_HISTORY_FILE) {
|
|
26
89
|
try {
|
|
@@ -29,38 +92,9 @@ export function recordMoaRun(record, filePath = DEFAULT_HISTORY_FILE) {
|
|
|
29
92
|
fs.mkdirSync(dir, { recursive: true })
|
|
30
93
|
}
|
|
31
94
|
|
|
32
|
-
|
|
33
|
-
id: record.id || crypto.randomUUID(),
|
|
34
|
-
timestamp: record.timestamp || new Date().toISOString(),
|
|
35
|
-
prompt: record.prompt || '',
|
|
36
|
-
preset: record.preset || 'default',
|
|
37
|
-
isRefinement: Boolean(record.isRefinement),
|
|
38
|
-
isFastMode: Boolean(record.isFastMode),
|
|
39
|
-
candidates: Array.isArray(record.candidates)
|
|
40
|
-
? record.candidates.map((c) => ({
|
|
41
|
-
provider: c.provider || '',
|
|
42
|
-
model: c.model || '',
|
|
43
|
-
filesCount: Array.isArray(c.files) ? c.files.length : (c.filesCount || 0),
|
|
44
|
-
usage: c.usage || { inputTokens: 0, outputTokens: 0 },
|
|
45
|
-
costUsd: Number(c.costUsd || 0),
|
|
46
|
-
}))
|
|
47
|
-
: [],
|
|
48
|
-
aggregator: record.aggregator
|
|
49
|
-
? {
|
|
50
|
-
provider: record.aggregator.provider || '',
|
|
51
|
-
model: record.aggregator.model || '',
|
|
52
|
-
usage: record.aggregator.usage || { inputTokens: 0, outputTokens: 0 },
|
|
53
|
-
costUsd: Number(record.aggregator.costUsd || 0),
|
|
54
|
-
}
|
|
55
|
-
: null,
|
|
56
|
-
winnerIndex: typeof record.winnerIndex === 'number' ? record.winnerIndex : -1,
|
|
57
|
-
winnerModel: record.winnerModel || '',
|
|
58
|
-
promotedFiles: Array.isArray(record.promotedFiles) ? record.promotedFiles : [],
|
|
59
|
-
totalTokens: Number(record.totalTokens || 0),
|
|
60
|
-
totalCostUsd: Number(record.totalCostUsd || 0),
|
|
61
|
-
durationMs: Number(record.durationMs || 0),
|
|
62
|
-
}
|
|
95
|
+
rotateHistoryFileIfNeeded(filePath)
|
|
63
96
|
|
|
97
|
+
const entry = normalizeRunRecord(record)
|
|
64
98
|
const line = JSON.stringify(entry) + '\n'
|
|
65
99
|
fs.appendFileSync(filePath, line, 'utf8')
|
|
66
100
|
return entry
|
|
@@ -70,6 +104,26 @@ export function recordMoaRun(record, filePath = DEFAULT_HISTORY_FILE) {
|
|
|
70
104
|
}
|
|
71
105
|
}
|
|
72
106
|
|
|
107
|
+
/**
|
|
108
|
+
* Asynchronously appends a completed MoA run record without blocking the event loop.
|
|
109
|
+
*/
|
|
110
|
+
export async function recordMoaRunAsync(record, filePath = DEFAULT_HISTORY_FILE) {
|
|
111
|
+
try {
|
|
112
|
+
const dir = path.dirname(filePath)
|
|
113
|
+
await fs.promises.mkdir(dir, { recursive: true }).catch(() => {})
|
|
114
|
+
|
|
115
|
+
rotateHistoryFileIfNeeded(filePath)
|
|
116
|
+
|
|
117
|
+
const entry = normalizeRunRecord(record)
|
|
118
|
+
const line = JSON.stringify(entry) + '\n'
|
|
119
|
+
await fs.promises.appendFile(filePath, line, 'utf8')
|
|
120
|
+
return entry
|
|
121
|
+
} catch (err) {
|
|
122
|
+
console.warn('[dsh-moa] Failed to async-record run to history:', err)
|
|
123
|
+
return null
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
73
127
|
/**
|
|
74
128
|
* Reads history runs with pagination (latest first).
|
|
75
129
|
*/
|
|
@@ -145,8 +199,6 @@ export function getMoaLeaderboard(filePath = DEFAULT_HISTORY_FILE) {
|
|
|
145
199
|
st.totalTokens += inTok + outTok
|
|
146
200
|
st.totalCostUsd += Number(cand.costUsd || 0)
|
|
147
201
|
|
|
148
|
-
// Exact match only: loose substring matching combined with empty
|
|
149
|
-
// candidate fields credited wins to every model in the run.
|
|
150
202
|
if (winner && cand.model && (winner === key || winner === cand.model)) {
|
|
151
203
|
st.wins += 1
|
|
152
204
|
}
|
package/lib/index.js
CHANGED
|
@@ -56,6 +56,9 @@ export const PresetSchema = z.object({
|
|
|
56
56
|
aggregator_fallbacks: z.array(ModelSlotSchema).default([]),
|
|
57
57
|
reference_temperature: z.number().default(0.6),
|
|
58
58
|
aggregator_temperature: z.number().default(0.4),
|
|
59
|
+
reference_timeout_sec: z.number().default(60),
|
|
60
|
+
aggregator_timeout_sec: z.number().default(180),
|
|
61
|
+
blind_evaluation: z.boolean().default(false),
|
|
59
62
|
max_tokens: z.number().default(4096),
|
|
60
63
|
judge_criteria: z.string().default(''),
|
|
61
64
|
})
|
|
@@ -156,7 +159,7 @@ export function apply(ctx, config) {
|
|
|
156
159
|
/**
|
|
157
160
|
* Unified LLM dispatch function using ctx.llm.prepareCall / stream
|
|
158
161
|
*/
|
|
159
|
-
const callLlm = async ({ provider, model, messages, temperature = 0.6, maxTokens = 4096, onStreamDelta }) => {
|
|
162
|
+
const callLlm = async ({ provider, model, messages, temperature = 0.6, maxTokens = 4096, timeoutMs, onStreamDelta }) => {
|
|
160
163
|
if (!ctx.llm) {
|
|
161
164
|
throw new Error('ctx.llm is not available in cordis context')
|
|
162
165
|
}
|
|
@@ -180,8 +183,9 @@ export function apply(ctx, config) {
|
|
|
180
183
|
throw new Error(`LLM provider/model ${provider}:${model} could not be prepared`)
|
|
181
184
|
}
|
|
182
185
|
|
|
186
|
+
const effectiveTimeoutMs = typeof timeoutMs === 'number' && timeoutMs > 0 ? timeoutMs : 120000
|
|
183
187
|
const abortCtrl = new AbortController()
|
|
184
|
-
const timeoutId = setTimeout(() => abortCtrl.abort(new Error(
|
|
188
|
+
const timeoutId = setTimeout(() => abortCtrl.abort(new Error(`LLM call timeout after ${Math.round(effectiveTimeoutMs / 1000)}s`)), effectiveTimeoutMs)
|
|
185
189
|
timeoutId.unref?.()
|
|
186
190
|
|
|
187
191
|
try {
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
// lib/moa-parser.js
|
|
2
|
+
// Parsers for model outputs, winner selection, commands, and code blocks for @goodandready/dsh-moa.
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Extracts candidate winner index from judge output.
|
|
6
|
+
* Looks for machine marker WINNER_CANDIDATE_INDEX: N first, then conversational fallbacks.
|
|
7
|
+
*/
|
|
8
|
+
export function parseWinnerIndex(judgeText, defaultIndex = 1, candidateCount = Infinity) {
|
|
9
|
+
if (!judgeText || typeof judgeText !== 'string') return defaultIndex
|
|
10
|
+
const inRange = (idx) => !isNaN(idx) && idx >= 1 && idx <= candidateCount
|
|
11
|
+
const match = /WINNER_CANDIDATE_INDEX:\s*(\d+)/i.exec(judgeText)
|
|
12
|
+
if (match) {
|
|
13
|
+
const idx = parseInt(match[1], 10)
|
|
14
|
+
if (inRange(idx)) return idx
|
|
15
|
+
}
|
|
16
|
+
const candMatch = /(?:Кандидат|Candidate|Reference)\s*(\d+)\b/i.exec(judgeText)
|
|
17
|
+
if (candMatch) {
|
|
18
|
+
const idx = parseInt(candMatch[1], 10)
|
|
19
|
+
if (inRange(idx)) return idx
|
|
20
|
+
}
|
|
21
|
+
return defaultIndex
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Parses the recommended assembler model index and label from curator output.
|
|
26
|
+
*/
|
|
27
|
+
export function parseRecommendedAssembler(judgeText, defaultIndex = 1, candidateCount = Infinity) {
|
|
28
|
+
if (!judgeText || typeof judgeText !== 'string') return { index: defaultIndex, label: '' }
|
|
29
|
+
const inRange = (idx) => !isNaN(idx) && idx >= 1 && idx <= candidateCount
|
|
30
|
+
const match = /RECOMMENDED_ASSEMBLER:\s*(\d+)(?:\s*\(([^)]+)\))?/i.exec(judgeText)
|
|
31
|
+
if (match) {
|
|
32
|
+
const idx = parseInt(match[1], 10)
|
|
33
|
+
if (inRange(idx)) {
|
|
34
|
+
return { index: idx, label: match[2]?.trim() || '' }
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
return { index: defaultIndex, label: '' }
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Parses a `/moa [preset] <prompt>` command invocation.
|
|
42
|
+
*/
|
|
43
|
+
export function parseMoACommand(text, presets = []) {
|
|
44
|
+
if (typeof text !== 'string' || !text.startsWith('/moa')) {
|
|
45
|
+
return null
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const remainder = text.slice(4).trim()
|
|
49
|
+
if (!remainder) {
|
|
50
|
+
return {
|
|
51
|
+
presetName: 'default',
|
|
52
|
+
prompt: '',
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const matchPreset = /^--preset(?:=|\s+)([a-zA-Z0-9_-]+)\s*(.*)/s.exec(remainder)
|
|
57
|
+
if (matchPreset) {
|
|
58
|
+
return {
|
|
59
|
+
presetName: matchPreset[1],
|
|
60
|
+
prompt: matchPreset[2] || '',
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const parts = remainder.split(/\s+/)
|
|
65
|
+
const firstWord = parts[0]
|
|
66
|
+
if (Array.isArray(presets)) {
|
|
67
|
+
const matched = presets.find((p) => p.name === firstWord)
|
|
68
|
+
if (matched) {
|
|
69
|
+
return {
|
|
70
|
+
presetName: matched.name,
|
|
71
|
+
prompt: parts.slice(1).join(' ').trim(),
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
presetName: 'default',
|
|
78
|
+
prompt: remainder,
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Replaces large code blocks with concise file/code summaries to avoid token waste and huge chat dumps.
|
|
84
|
+
*/
|
|
85
|
+
export function stripOrSummarizeCode(text) {
|
|
86
|
+
if (!text || typeof text !== 'string') return ''
|
|
87
|
+
return text.replace(/```([a-zA-Z0-9_\-\.\/]*)\s*([\w\.\/\-]+\.[a-zA-Z0-9]+)?\n([\s\S]*?)```/g, (match, lang, fileTag, code) => {
|
|
88
|
+
const lines = code.trim().split('\n')
|
|
89
|
+
if (lines.length <= 3 && !/html|jsx|tsx|vue|svelte|css|js|ts/i.test(lang)) {
|
|
90
|
+
return match
|
|
91
|
+
}
|
|
92
|
+
const fileHint = fileTag || (code.match(/^\s*(?:\/\/|#|<!--|\/\*)\s*(?:file|filepath|path):\s*([^\s*]+)/im)?.[1])
|
|
93
|
+
const label = fileHint ? `file \`${fileHint}\`` : (lang ? `code \`${lang}\`` : 'code')
|
|
94
|
+
return `\n> 📄 *[${label} - ${lines.length} lines saved to disk]*\n`
|
|
95
|
+
})
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Formats user-facing markdown response for synthesis, questions, and failure kinds.
|
|
100
|
+
*/
|
|
101
|
+
export function formatMoAResponse({ moaResult, presetName }) {
|
|
102
|
+
const parts = []
|
|
103
|
+
const pName = presetName || moaResult?.presetName || 'default'
|
|
104
|
+
const judge = moaResult?.aggregator || 'unknown'
|
|
105
|
+
const refs = moaResult?.references || []
|
|
106
|
+
|
|
107
|
+
if (moaResult?.kind === 'questions') {
|
|
108
|
+
parts.push('## 🧠 Mixture of Agents — Requirements Clarification')
|
|
109
|
+
parts.push(`*Judge (${judge}) and the advisors analyzed the task:*\n`)
|
|
110
|
+
parts.push(moaResult.content)
|
|
111
|
+
return parts.join('\n')
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
if (moaResult?.kind === 'failure') {
|
|
115
|
+
parts.push('## ⚠️ Mixture of Agents — Execution Failed')
|
|
116
|
+
parts.push(moaResult.content)
|
|
117
|
+
return parts.join('\n')
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const modeBadge = moaResult?.isFastMode
|
|
121
|
+
? '⚡ Fast Mode'
|
|
122
|
+
: (moaResult?.isCuratorSynthesis ? `🧠 Curator: ${judge}` : `Judge: ${judge}`)
|
|
123
|
+
parts.push(`## 🧠 Mixture of Agents (Preset: ${pName} | ${modeBadge})`)
|
|
124
|
+
parts.push('')
|
|
125
|
+
|
|
126
|
+
if (moaResult?.isRefinement) {
|
|
127
|
+
parts.push('> 🔄 **Mode**: Iterative project refinement (Refinement)')
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const hasPromoted = moaResult?.promotedFiles && moaResult.promotedFiles.length > 0
|
|
131
|
+
if (hasPromoted) {
|
|
132
|
+
parts.push(`> 📦 **Files created in the project**: \`${moaResult.promotedFiles.join('`, `')}\``)
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
if (moaResult?.recommendedAssembler?.label) {
|
|
136
|
+
parts.push(`> 🎯 **Recommended Master Assembler**: Candidate ${moaResult.recommendedAssembler.index} (\`${moaResult.recommendedAssembler.label}\`)`)
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
if (moaResult?.liveCanvas?.previewUrl) {
|
|
140
|
+
parts.push(`> 🎨 **Live Canvas**: [🚀 Открыть ${moaResult.liveCanvas.title || 'превью'} в Live Canvas](${moaResult.liveCanvas.previewUrl}) | [↗ Открыть в новой вкладке](${moaResult.liveCanvas.previewUrl})`)
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// Cost tracking card
|
|
144
|
+
if (moaResult?.usage) {
|
|
145
|
+
const u = moaResult.usage
|
|
146
|
+
const costStr = u.totalCostUsd > 0 ? `~\$${u.totalCostUsd.toFixed(4)}` : 'Free'
|
|
147
|
+
const tokStr = u.totalTokens >= 1000 ? `${(u.totalTokens / 1000).toFixed(1)}k` : `${u.totalTokens}`
|
|
148
|
+
parts.push(`> 💰 **Run cost**: ${costStr} (${tokStr} tokens total)`)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
parts.push('')
|
|
152
|
+
|
|
153
|
+
if (!moaResult?.isFastMode) {
|
|
154
|
+
parts.push(`### ⚖️ Judge verdict and final synthesis (Synthesis: ${judge})`)
|
|
155
|
+
parts.push('')
|
|
156
|
+
const cleanJudgeContent = hasPromoted
|
|
157
|
+
? stripOrSummarizeCode(moaResult?.content || '')
|
|
158
|
+
: (moaResult?.content || '(нет ответа)')
|
|
159
|
+
parts.push(cleanJudgeContent)
|
|
160
|
+
parts.push('')
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
if (Array.isArray(refs) && refs.length > 0) {
|
|
164
|
+
const title = moaResult?.isFastMode ? '### 🚀 Candidate generation result:' : `### 👥 Advisor responses (${refs.length}):`
|
|
165
|
+
parts.push(title)
|
|
166
|
+
parts.push('')
|
|
167
|
+
refs.forEach((ref, i) => {
|
|
168
|
+
const statusIcon = ref.ok ? '✅' : '⚠️'
|
|
169
|
+
const fileBadge = ref.files?.length ? ` (${ref.files.length} файл(ов))` : ''
|
|
170
|
+
const costBadge = ref.costUsd > 0 ? ` [~\$${ref.costUsd.toFixed(4)}]` : ''
|
|
171
|
+
parts.push(`#### ${statusIcon} Model ${i + 1}: ${ref.label}${fileBadge}${costBadge}`)
|
|
172
|
+
parts.push('')
|
|
173
|
+
parts.push(stripOrSummarizeCode(ref.text))
|
|
174
|
+
parts.push('')
|
|
175
|
+
parts.push('---')
|
|
176
|
+
parts.push('')
|
|
177
|
+
})
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
return parts.join('\n')
|
|
181
|
+
}
|