@ludi-uni/ludi-agent-kit 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +55 -0
- package/LICENSE +21 -0
- package/README.md +107 -0
- package/adapters/codex/README.md +24 -0
- package/adapters/codex/skill-metadata/visual-verification/agents/openai.yaml +7 -0
- package/adapters/pi/README.md +88 -0
- package/adapters/pi/browser/agent-browser.mjs +193 -0
- package/adapters/pi/lib/invoke.mjs +55 -0
- package/adapters/pi/lib/list-models.mjs +29 -0
- package/adapters/pi/lib/settings-proposal.mjs +34 -0
- package/adapters/pi/lib/subagent.mjs +175 -0
- package/adapters/pi/loop-guard/index.js +51 -0
- package/adapters/pi/maintenance-policy.json +36 -0
- package/adapters/pi/mcp.template.json +4 -0
- package/adapters/pi/model-catalog.json +97 -0
- package/adapters/pi/models.json +13 -0
- package/adapters/pi/models.local.example.json +14 -0
- package/adapters/pi/orchestrator-ext/command.mjs +14 -0
- package/adapters/pi/orchestrator-ext/index.js +150 -0
- package/adapters/pi/settings.template.json +7 -0
- package/adapters/pi/shell-gate/index.js +70 -0
- package/adapters/pi/sync-pi.ps1 +137 -0
- package/agents/README.md +26 -0
- package/agents/browser.md +64 -0
- package/agents/coder.md +31 -0
- package/agents/orchestrator.md +37 -0
- package/agents/reviewer.md +32 -0
- package/agents/scout.md +35 -0
- package/agents/tester.md +28 -0
- package/agents/visual.md +28 -0
- package/context-pack/SPEC.md +101 -0
- package/context-pack/context-pack.schema.json +79 -0
- package/context-pack/examples/example-fix.md +44 -0
- package/docs/architecture.md +55 -0
- package/docs/migration-from-codex-setting.md +44 -0
- package/docs/model-maintenance.md +401 -0
- package/docs/orchestrator.md +155 -0
- package/docs/phase2-report.md +39 -0
- package/docs/roadmap.md +27 -0
- package/docs/third-party.md +15 -0
- package/lib/agents.mjs +79 -0
- package/lib/context-pack.mjs +215 -0
- package/lib/job.mjs +312 -0
- package/lib/language-policy.mjs +27 -0
- package/lib/maintenance-exec.mjs +377 -0
- package/lib/maintenance-runner.mjs +266 -0
- package/lib/maintenance.mjs +422 -0
- package/lib/normalize.mjs +101 -0
- package/lib/observe/differ.mjs +185 -0
- package/lib/observe/observation.mjs +147 -0
- package/lib/observe/observers.mjs +134 -0
- package/lib/observe/sources.mjs +154 -0
- package/lib/orchestrator/activity.mjs +249 -0
- package/lib/orchestrator/api.mjs +151 -0
- package/lib/orchestrator/contract.mjs +68 -0
- package/lib/orchestrator/escalation.mjs +84 -0
- package/lib/orchestrator/evaluator.mjs +92 -0
- package/lib/orchestrator/failures.mjs +88 -0
- package/lib/orchestrator/health.mjs +53 -0
- package/lib/orchestrator/orchestrator.mjs +483 -0
- package/lib/orchestrator/permissions.mjs +64 -0
- package/lib/orchestrator/planner.mjs +194 -0
- package/lib/orchestrator/policy.mjs +134 -0
- package/lib/orchestrator/router.mjs +45 -0
- package/lib/orchestrator/runner.mjs +278 -0
- package/lib/orchestrator/shell-policy.mjs +52 -0
- package/lib/orchestrator/store.mjs +581 -0
- package/lib/orchestrator/task-store.mjs +79 -0
- package/lib/orchestrator/turn-budget.mjs +63 -0
- package/lib/orchestrator/worktree.mjs +72 -0
- package/lib/pipeline.mjs +279 -0
- package/lib/registry.mjs +63 -0
- package/lib/resolve.mjs +35 -0
- package/lib/routing.mjs +137 -0
- package/lib/telemetry.mjs +222 -0
- package/mcp/README.md +11 -0
- package/mcp/servers.json +13 -0
- package/orchestration/decision-policy.json +66 -0
- package/package.json +56 -0
- package/routing/README.md +24 -0
- package/routing/routing.json +81 -0
- package/routing/routing.schema.json +66 -0
- package/rules/README.md +10 -0
- package/rules/common.md +52 -0
- package/rules/loop-prevention.md +15 -0
- package/rules/repo-local.md +6 -0
- package/scripts/check-environment.ps1 +22 -0
- package/scripts/context-pack.mjs +17 -0
- package/scripts/e2e-investigate-repro.mjs +66 -0
- package/scripts/model-maintenance-job.mjs +59 -0
- package/scripts/observe-models.mjs +97 -0
- package/scripts/orchestrate.mjs +137 -0
- package/scripts/reevaluate-models.mjs +95 -0
- package/scripts/report-model-maintenance.mjs +70 -0
- package/scripts/resolve-capabilities.mjs +39 -0
- package/scripts/run-pipeline.mjs +56 -0
- package/scripts/sync-agents-md.ps1 +10 -0
- package/scripts/validate.mjs +71 -0
- package/skills/README.md +14 -0
- package/skills/pi-workflow/SKILL.md +26 -0
- package/skills/pi-workflow/references/code-investigation-and-fix.md +16 -0
- package/skills/pi-workflow/references/research.md +14 -0
- package/skills/pi-workflow/references/review.md +11 -0
- package/skills/pi-workflow/references/visual-work.md +14 -0
- package/skills/project-management/SKILL.md +106 -0
- package/skills/project-management/references/operations.md +52 -0
- package/skills/visual-verification/SKILL.md +88 -0
- package/skills/visual-verification/scripts/analyze-speech.ps1 +346 -0
- package/skills/visual-verification/scripts/backends/whisperx_backend.py +234 -0
- package/skills/visual-verification/scripts/common.ps1 +387 -0
- package/skills/visual-verification/scripts/contact-sheet.ps1 +121 -0
- package/skills/visual-verification/scripts/desktop-discover.ps1 +45 -0
- package/skills/visual-verification/scripts/desktop-inspect.ps1 +67 -0
- package/skills/visual-verification/scripts/desktop-record.ps1 +97 -0
- package/skills/visual-verification/scripts/desktop-screenshot.ps1 +65 -0
- package/skills/visual-verification/scripts/evaluate-sync.ps1 +249 -0
- package/skills/visual-verification/scripts/extract-frames.ps1 +79 -0
- package/skills/visual-verification/scripts/inspect-media.ps1 +138 -0
- package/skills/visual-verification/scripts/record-av.ps1 +102 -0
- package/skills/visual-verification/scripts/record.ps1 +72 -0
- package/skills/visual-verification/scripts/screenshot.ps1 +44 -0
- package/skills/visual-verification/scripts/waveform.ps1 +450 -0
- package/skills/visual-verification/scripts/winapp-common.ps1 +465 -0
- package/tests/activity.test.mjs +252 -0
- package/tests/attempt-budget.test.mjs +102 -0
- package/tests/browser.test.mjs +121 -0
- package/tests/context-pack.test.mjs +98 -0
- package/tests/dirty-gate.test.mjs +211 -0
- package/tests/e2e-browser.mjs +66 -0
- package/tests/e2e-real-orchestrator-resume.mjs +101 -0
- package/tests/e2e-real-orchestrator.mjs +41 -0
- package/tests/e2e-real-pi.mjs +27 -0
- package/tests/e2e-real-tool-orchestrator.mjs +66 -0
- package/tests/fixtures/browser-page/index.html +20 -0
- package/tests/fixtures/maintenance/availability.txt +5 -0
- package/tests/fixtures/maintenance/catalog.json +74 -0
- package/tests/fixtures/maintenance/events.json +13 -0
- package/tests/fixtures/math-repo/README.md +3 -0
- package/tests/fixtures/math-repo/package.json +7 -0
- package/tests/fixtures/math-repo/src/math.js +11 -0
- package/tests/fixtures/math-repo/test/math.test.js +7 -0
- package/tests/fixtures/observe/announcements.json +8 -0
- package/tests/fixtures/orch-concurrent-child.mjs +44 -0
- package/tests/fixtures/orch-persist-child.mjs +61 -0
- package/tests/job.test.mjs +230 -0
- package/tests/kit.test.mjs +79 -0
- package/tests/language-policy.test.mjs +93 -0
- package/tests/loop-guard.test.mjs +60 -0
- package/tests/maintenance-exec.test.mjs +218 -0
- package/tests/maintenance-runner.test.mjs +222 -0
- package/tests/maintenance.test.mjs +195 -0
- package/tests/observe.test.mjs +283 -0
- package/tests/observer-registry.test.mjs +157 -0
- package/tests/orchestrator-cleanup.test.mjs +358 -0
- package/tests/orchestrator-command.test.mjs +14 -0
- package/tests/orchestrator-persist.test.mjs +375 -0
- package/tests/orchestrator-tools.test.mjs +215 -0
- package/tests/orchestrator.test.mjs +396 -0
- package/tests/package.test.mjs +37 -0
- package/tests/pipeline.test.mjs +239 -0
- package/tests/planner-classification.test.mjs +81 -0
- package/tests/planner-split.test.mjs +67 -0
- package/tests/qoder-observer.test.mjs +266 -0
- package/tests/reassign-progression.test.mjs +104 -0
- package/tests/retry-escalation.test.mjs +120 -0
- package/tests/routing.test.mjs +110 -0
- package/tests/sqlite-concurrency.test.mjs +178 -0
- package/tests/task-global-e2e.test.mjs +63 -0
- package/tests/task-global-failed.test.mjs +134 -0
- package/tests/telemetry.test.mjs +173 -0
- package/tests/test-sync-pi.ps1 +56 -0
- package/tests/turn-budget.test.mjs +106 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
// Subagent execution budget: complexity classification, per-role initial turn
|
|
2
|
+
// budgets, and deterministic progress evaluation for bounded extension.
|
|
3
|
+
// Provider-agnostic — nothing here names a model or backend.
|
|
4
|
+
|
|
5
|
+
export const COMPLEXITIES = ['simple', 'normal', 'heavy', 'repo-history-heavy'];
|
|
6
|
+
|
|
7
|
+
const HEAVY_RE = /\b(git log|commit history|commit(s)? histor|across (the )?(repo|codebase)|multiple (directories|files|modules)|whole (repo|codebase)|all files|every file)\b|コミット履歴|履歴|横断|全体/i;
|
|
8
|
+
const HISTORY_RE = /\b(git|commit|history|changelog|blame|past changes|previous commits)\b|コミット|変更履歴|過去/i;
|
|
9
|
+
const MULTI_RE = /\b(and|&|plus|also|then combine|integrate|merge|compare|across)\b|かつ|と同時に|統合|比較|組み合わせ/i;
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Classify a task's investigation complexity from its goal/kind. Deterministic —
|
|
13
|
+
* no model call. repo-history-heavy needs git/commit-history analysis; heavy needs
|
|
14
|
+
* multi-area discovery+analysis; normal is a focused investigate; simple is trivial.
|
|
15
|
+
*/
|
|
16
|
+
export function classifyTaskComplexity(task) {
|
|
17
|
+
const text = `${task?.title ?? ''} ${task?.goal ?? ''}`.toLowerCase();
|
|
18
|
+
const deps = task?.dependencies?.length ?? 0;
|
|
19
|
+
const history = HISTORY_RE.test(text);
|
|
20
|
+
const heavy = HEAVY_RE.test(text) || MULTI_RE.test(text);
|
|
21
|
+
if (history && (heavy || /ux|preference|ui|implement|design|pattern/i.test(text))) return 'repo-history-heavy';
|
|
22
|
+
if (history) return 'repo-history-heavy';
|
|
23
|
+
if (heavy || deps >= 2) return 'heavy';
|
|
24
|
+
if (task?.kind === 'investigate' || task?.kind === 'review') return 'normal';
|
|
25
|
+
return 'simple';
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Initial turn budget for a task = turn_budgets[role][complexity]
|
|
30
|
+
* -> turn_budgets[role].normal -> turn_budgets.default[complexity] -> max_turns.
|
|
31
|
+
*/
|
|
32
|
+
export function initialTurnBudget(agentRuntime, role, complexity) {
|
|
33
|
+
const budgets = agentRuntime?.turn_budgets ?? {};
|
|
34
|
+
const roleB = budgets[role] ?? budgets.default ?? {};
|
|
35
|
+
const defB = budgets.default ?? {};
|
|
36
|
+
return roleB[complexity] ?? roleB.normal ?? defB[complexity] ?? defB.normal ?? agentRuntime?.max_turns ?? 12;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Deterministic progress score from a subagent's execution telemetry.
|
|
41
|
+
* Positive score = meaningful progress worth extending. No LLM judgement.
|
|
42
|
+
* +2 per tool call, +3 per unique file inspected, +1 per successful command,
|
|
43
|
+
* -2 per repeated identical command, -4 for zero tool calls.
|
|
44
|
+
*/
|
|
45
|
+
export function progressScore(telemetry) {
|
|
46
|
+
const toolCalls = telemetry?.toolCalls ?? 0;
|
|
47
|
+
const uniqueFiles = telemetry?.uniqueFiles?.size ?? telemetry?.uniqueFilesInspected ?? 0;
|
|
48
|
+
const commands = telemetry?.commands ?? [];
|
|
49
|
+
const successful = telemetry?.successfulToolCalls ?? commands.length;
|
|
50
|
+
const repeats = commands.length - new Set(commands).size;
|
|
51
|
+
const score = toolCalls * 2 + uniqueFiles * 3 + successful * 1 - repeats * 2 + (toolCalls === 0 ? -4 : 0);
|
|
52
|
+
const reasons = [];
|
|
53
|
+
if (toolCalls === 0) reasons.push('no tool calls');
|
|
54
|
+
if (uniqueFiles > 0) reasons.push(`${uniqueFiles} files inspected`);
|
|
55
|
+
if (repeats > 1) reasons.push(`${repeats} repeated commands`);
|
|
56
|
+
if (successful > 0) reasons.push(`${successful} commands run`);
|
|
57
|
+
return { score, reasons, meaningful: toolCalls > 0 && (uniqueFiles > 0 || successful > 0 || repeats <= 1) };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** True when a turn-limit hit should be treated as no-progress (global model failure). */
|
|
61
|
+
export function isNoProgressTimeout(telemetry) {
|
|
62
|
+
return !progressScore(telemetry).meaningful;
|
|
63
|
+
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
// Baseline the worktree before a task and report only what changed. Never reset, checkout, or stash.
|
|
2
|
+
import { createHash } from 'node:crypto';
|
|
3
|
+
import { spawnSync } from 'node:child_process';
|
|
4
|
+
import { readFileSync, readdirSync, statSync, existsSync } from 'node:fs';
|
|
5
|
+
import { join, relative } from 'node:path';
|
|
6
|
+
|
|
7
|
+
const SKIP = new Set(['.git', 'node_modules', 'out', 'dist', '.orchestration']);
|
|
8
|
+
|
|
9
|
+
function hashFile(path) {
|
|
10
|
+
try { return createHash('sha256').update(readFileSync(path)).digest('hex'); }
|
|
11
|
+
catch { return null; }
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
function parsePorcelain(stdout) {
|
|
15
|
+
const parts = String(stdout ?? '').split('\0').filter(Boolean);
|
|
16
|
+
const entries = {};
|
|
17
|
+
for (let i = 0; i < parts.length; i++) {
|
|
18
|
+
const rec = parts[i];
|
|
19
|
+
const code = rec.slice(0, 2);
|
|
20
|
+
let file = rec.slice(3);
|
|
21
|
+
if ((code.startsWith('R') || code.startsWith('C')) && parts[i + 1]) file = parts[++i];
|
|
22
|
+
entries[file.replace(/\\/g, '/')] = code;
|
|
23
|
+
}
|
|
24
|
+
return entries;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function inventory(root, dir = root, out = {}, budget = { n: 0 }) {
|
|
28
|
+
if (budget.n > 400) return out;
|
|
29
|
+
let names = [];
|
|
30
|
+
try { names = readdirSync(dir); } catch { return out; }
|
|
31
|
+
for (const name of names) {
|
|
32
|
+
if (SKIP.has(name)) continue;
|
|
33
|
+
const abs = join(dir, name);
|
|
34
|
+
let st;
|
|
35
|
+
try { st = statSync(abs); } catch { continue; }
|
|
36
|
+
if (st.isDirectory()) inventory(root, abs, out, budget);
|
|
37
|
+
else if (st.isFile()) {
|
|
38
|
+
budget.n++;
|
|
39
|
+
out[relative(root, abs).replace(/\\/g, '/')] = hashFile(abs);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
return out;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function captureWorktree(root) {
|
|
46
|
+
if (!root || !existsSync(root)) return { available: false, entries: {} };
|
|
47
|
+
const status = spawnSync('git', ['status', '--porcelain', '-z'], { cwd: root, encoding: 'utf8', windowsHide: true });
|
|
48
|
+
if (status.error || status.status !== 0) return { available: false, entries: inventory(root), source: 'files' };
|
|
49
|
+
const codes = parsePorcelain(status.stdout);
|
|
50
|
+
const entries = {};
|
|
51
|
+
for (const [file, code] of Object.entries(codes)) entries[file] = { code, hash: hashFile(join(root, file)) };
|
|
52
|
+
return { available: true, entries, source: 'git' };
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function diffWorktree(before, after) {
|
|
56
|
+
if (!before?.source || before.source !== after?.source) return [];
|
|
57
|
+
const changed = [];
|
|
58
|
+
const keys = new Set([...Object.keys(before.entries ?? {}), ...Object.keys(after.entries ?? {})]);
|
|
59
|
+
for (const path of keys) {
|
|
60
|
+
const b = before.entries[path];
|
|
61
|
+
const a = after.entries[path];
|
|
62
|
+
if (JSON.stringify(b ?? null) === JSON.stringify(a ?? null)) continue;
|
|
63
|
+
const code = String(a?.code ?? '');
|
|
64
|
+
let kind = 'modified';
|
|
65
|
+
if (!a) kind = 'deleted';
|
|
66
|
+
else if (after.source === 'git' && code.includes('?')) kind = 'untracked';
|
|
67
|
+
else if (!b || (after.source === 'git' && code.includes('A'))) kind = 'added';
|
|
68
|
+
else if (after.source === 'git' && code.includes('D')) kind = 'deleted';
|
|
69
|
+
changed.push({ path, kind, before: b?.code ?? null, after: a?.code ?? null });
|
|
70
|
+
}
|
|
71
|
+
return changed;
|
|
72
|
+
}
|
package/lib/pipeline.mjs
ADDED
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
// Minimal executable path: task -> scout -> Context Pack -> coder -> validation -> success | escalation.
|
|
2
|
+
// Adapter-agnostic: the caller supplies `invoke({ modelId, provider, model, thinking, systemPrompt, prompt, cwd })`
|
|
3
|
+
// returning { ok, text, error?, durationMs }. Nothing here knows provider names.
|
|
4
|
+
import { readFileSync, writeFileSync, existsSync, readdirSync, statSync, mkdirSync } from 'node:fs';
|
|
5
|
+
import { join, relative, resolve, dirname } from 'node:path';
|
|
6
|
+
import { spawnSync } from 'node:child_process';
|
|
7
|
+
import { resolveCapability } from './resolve.mjs';
|
|
8
|
+
import { LANGUAGE_POLICY } from './language-policy.mjs';
|
|
9
|
+
import { normalizeContextPack } from './normalize.mjs';
|
|
10
|
+
import { toMarkdown } from './context-pack.mjs';
|
|
11
|
+
|
|
12
|
+
export const DEFAULT_MAX_ATTEMPTS = 2;
|
|
13
|
+
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'out', 'dist', '.pi']);
|
|
14
|
+
|
|
15
|
+
// ---------- repository survey (deterministic pre-pass; keeps the scout prompt small) ----------
|
|
16
|
+
export function surveyRepo(repoRoot, { maxFiles = 200, maxInlineBytes = 4000, maxTotalInline = 24000 } = {}) {
|
|
17
|
+
const files = [];
|
|
18
|
+
const walk = dir => {
|
|
19
|
+
for (const e of readdirSync(dir, { withFileTypes: true })) {
|
|
20
|
+
if (IGNORE_DIRS.has(e.name)) continue;
|
|
21
|
+
const p = join(dir, e.name);
|
|
22
|
+
if (e.isDirectory()) walk(p); else files.push(p);
|
|
23
|
+
if (files.length >= maxFiles) return;
|
|
24
|
+
}
|
|
25
|
+
};
|
|
26
|
+
walk(repoRoot);
|
|
27
|
+
const rel = files.map(f => ({ path: relative(repoRoot, f).replace(/\\/g, '/'), size: statSync(f).size }));
|
|
28
|
+
let budget = maxTotalInline;
|
|
29
|
+
const inline = [];
|
|
30
|
+
for (const f of rel) {
|
|
31
|
+
if (f.size > maxInlineBytes || budget <= 0 || /\.(png|jpg|mp4|wav|lock)$/.test(f.path)) continue;
|
|
32
|
+
const content = readFileSync(join(repoRoot, f.path), 'utf8');
|
|
33
|
+
inline.push({ path: f.path, content });
|
|
34
|
+
budget -= f.size;
|
|
35
|
+
}
|
|
36
|
+
return { files: rel, inline };
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function detectTestCommand(repoRoot) {
|
|
40
|
+
const pkg = join(repoRoot, 'package.json');
|
|
41
|
+
if (existsSync(pkg)) {
|
|
42
|
+
const json = JSON.parse(readFileSync(pkg, 'utf8'));
|
|
43
|
+
if (json.scripts?.test) return 'npm test';
|
|
44
|
+
}
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function runTests(repoRoot, command) {
|
|
49
|
+
if (!command) return { ok: false, skipped: true, output: 'no test command' };
|
|
50
|
+
const shell = process.platform === 'win32' ? ['pwsh', ['-NoProfile', '-Command', command]] : ['sh', ['-c', command]];
|
|
51
|
+
// Strip node test-runner context so a nested `node --test` reports to us, not to an outer runner.
|
|
52
|
+
const env = Object.fromEntries(Object.entries(process.env).filter(([k]) => !k.startsWith('NODE_TEST_') && k !== 'NODE_OPTIONS'));
|
|
53
|
+
const r = spawnSync(shell[0], shell[1], { cwd: repoRoot, encoding: 'utf8', timeout: 120000, windowsHide: true, env });
|
|
54
|
+
return { ok: r.status === 0, status: r.status, output: ((r.stdout ?? '') + (r.stderr ?? '')).slice(-6000) };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// ---------- prompts ----------
|
|
58
|
+
export function scoutPrompt({ task, survey, testOutput, testCommand }) {
|
|
59
|
+
const tree = survey.files.map(f => `- ${f.path} (${f.size} B)`).join('\n');
|
|
60
|
+
const contents = survey.inline.map(f => `### \`${f.path}\`\n\`\`\`\n${f.content}\n\`\`\``).join('\n\n');
|
|
61
|
+
return [
|
|
62
|
+
`Task: ${task}`,
|
|
63
|
+
'', '## Repository files', tree,
|
|
64
|
+
'', '## File contents (bounded)', contents,
|
|
65
|
+
testOutput ? `\n## Failing test output\n\`\`\`\n${testOutput}\n\`\`\`` : '',
|
|
66
|
+
testCommand ? `\n## Test command\n- \`${testCommand}\`` : '',
|
|
67
|
+
'', 'Produce the Context Pack now. Start with the line `# Context Pack`. If the task needs files that do not exist yet,',
|
|
68
|
+
'list them in `## relevant_files` with the reason prefixed by `(new)`. If no existing file is relevant, add',
|
|
69
|
+
'`## discovery` containing `none — <why>`. Use repository-relative paths only.',
|
|
70
|
+
'', 'Write natural-language content (goal, reasons, constraints, notes) in Japanese unless the user requested another language.',
|
|
71
|
+
'CRITICAL: `##` section headings must be the EXACT schema names — `## task`, `## goal`, `## constraints`, `## relevant_files`, `## relevant_snippets`, `## repo_rules`, `## observed_errors`, `## test_commands`, `## previous_attempts`, `## expected_output`. Never translate or annotate a heading (no "## 課題", no "## task Japanese"). Only the prose under each heading is Japanese.',
|
|
72
|
+
].join('\n');
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function coderPrompt({ pack, repoRoot }) {
|
|
76
|
+
const files = pack.relevant_files.map(f => {
|
|
77
|
+
const p = join(repoRoot, f.path);
|
|
78
|
+
return existsSync(p) ? `### \`${f.path}\`\n\`\`\`\n${readFileSync(p, 'utf8')}\n\`\`\`` : `### \`${f.path}\` (does not exist yet${f.create ? '; you may create it' : ''})`;
|
|
79
|
+
}).join('\n\n');
|
|
80
|
+
return [
|
|
81
|
+
toMarkdown(pack),
|
|
82
|
+
'', '## Current contents of relevant_files', files,
|
|
83
|
+
'', 'Return the complete new contents of every file you change using exactly this format and nothing else:',
|
|
84
|
+
'=== FILE: <repo-relative path> ===', '<full file content>', '=== END ===',
|
|
85
|
+
'Only paths listed in relevant_files may be changed or created. Do not explain.',
|
|
86
|
+
'', 'Any prose you must include (commit-style notes, comments) should be in Japanese unless the user requested another language; code and file contents stay as-is.',
|
|
87
|
+
].join('\n');
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function parseFileBlocks(text) {
|
|
91
|
+
const blocks = [];
|
|
92
|
+
const re = /=== FILE: (.+?) ===\r?\n([\s\S]*?)\r?\n=== END ===/g;
|
|
93
|
+
let m;
|
|
94
|
+
while ((m = re.exec(text))) blocks.push({ path: m[1].trim().replace(/\\/g, '/'), content: m[2] });
|
|
95
|
+
return blocks;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export function applyFileBlocks(repoRoot, blocks, allowed) {
|
|
99
|
+
const applied = [], rejected = [];
|
|
100
|
+
const allow = new Set(allowed.map(f => f.path));
|
|
101
|
+
for (const b of blocks) {
|
|
102
|
+
if (!allow.has(b.path) || b.path.includes('..')) { rejected.push(b.path); continue; }
|
|
103
|
+
const target = resolve(repoRoot, b.path);
|
|
104
|
+
if (!target.startsWith(resolve(repoRoot))) { rejected.push(b.path); continue; }
|
|
105
|
+
mkdirSync(dirname(target), { recursive: true });
|
|
106
|
+
writeFileSync(target, b.content.endsWith('\n') ? b.content : b.content + '\n');
|
|
107
|
+
applied.push(b.path);
|
|
108
|
+
}
|
|
109
|
+
return { applied, rejected };
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// ---------- escalation core ----------
|
|
113
|
+
/**
|
|
114
|
+
* Try `fn(candidate)` on the capability chain: primary, then fallback[0], ... bounded by maxAttempts.
|
|
115
|
+
* The same modelId is never tried twice. Each failure is appended to pack.previous_attempts.
|
|
116
|
+
* Optional hooks: `skip(candidate)` returns a reason to pass over a candidate without spending an attempt
|
|
117
|
+
* (e.g. a backend known to be out of quota); `onFailure(candidate, reason)` observes each failed attempt.
|
|
118
|
+
*/
|
|
119
|
+
export async function withEscalation({ routing, registry, capability, agent, pack, maxAttempts = DEFAULT_MAX_ATTEMPTS, trace, fn, skip = null, onFailure = null, excludeModels = null, taskGlobalFailedModels = null, invocationsBudget = null }) {
|
|
120
|
+
const resolved = resolveCapability(routing, registry, capability);
|
|
121
|
+
if (!resolved.candidates.length) throw new Error(`no bound model for capability "${capability}" (unbound=${resolved.unbound.join(',')}, placeholder=${resolved.placeholder.join(',')})`);
|
|
122
|
+
// Candidates already tried (and failed) on earlier attempts of this task are
|
|
123
|
+
// passed in via excludeModels so a retried task advances to a NEW candidate
|
|
124
|
+
// instead of restarting at candidate[0]. They are recorded as skipped, not run.
|
|
125
|
+
//
|
|
126
|
+
// Counter semantics (attempt budget applies to invocationsStarted only):
|
|
127
|
+
// candidatesConsidered = every candidate the loop looked at
|
|
128
|
+
// candidatesSkipped = health/availability/already-tried skips (NO invocation)
|
|
129
|
+
// invocationsStarted = candidates actually invoked (attempt budget unit)
|
|
130
|
+
// invokedModels = modelIds actually invoked this call (NOT excludeModels)
|
|
131
|
+
const alreadyTried = new Set(excludeModels ?? []);
|
|
132
|
+
const globallyFailed = new Set(taskGlobalFailedModels ?? []);
|
|
133
|
+
const invokedSet = new Set(); // dedup within this call: identical modelId on two backends runs once
|
|
134
|
+
const invokedModels = [];
|
|
135
|
+
const skipped = [];
|
|
136
|
+
let last = null, attempts = 0, i = 0, considered = 0;
|
|
137
|
+
// The capability is freshly resolved each call. Filter task-global failures
|
|
138
|
+
// before the first invocation, including candidates after an early success.
|
|
139
|
+
const eligible = resolved.candidates.filter(c => {
|
|
140
|
+
if (!globallyFailed.has(c.modelId)) return true;
|
|
141
|
+
considered++;
|
|
142
|
+
skipped.push({ backend: c.backend, modelId: c.modelId, reason: 'task-global-failed' });
|
|
143
|
+
trace.push({ step: agent, agent, capability, backend: c.backend, provider: c.provider, model: c.model, modelId: c.modelId, skipped: true, ok: false, reason: 'task-global-failed', startedAt: new Date().toISOString(), durationMs: 0 });
|
|
144
|
+
return false;
|
|
145
|
+
});
|
|
146
|
+
// invocationsBudget caps ACTUAL invocations across the whole task (not just this
|
|
147
|
+
// call), so a retried task cannot overspend max_total_attempts_per_task.
|
|
148
|
+
const invocationsLeft = () => (invocationsBudget == null ? Infinity : invocationsBudget - invokedModels.length);
|
|
149
|
+
for (; i < eligible.length && attempts < maxAttempts && invocationsLeft() > 0; i++) {
|
|
150
|
+
const c = eligible[i];
|
|
151
|
+
considered++;
|
|
152
|
+
if (alreadyTried.has(c.modelId) || invokedSet.has(c.modelId)) { const reason = 'already-tried'; skipped.push({ backend: c.backend, modelId: c.modelId, reason }); trace.push({ step: agent, agent, capability, backend: c.backend, provider: c.provider, model: c.model, modelId: c.modelId, skipped: true, ok: false, reason, startedAt: new Date().toISOString(), durationMs: 0 }); continue; }
|
|
153
|
+
const why = skip?.(c);
|
|
154
|
+
if (why) {
|
|
155
|
+
skipped.push({ backend: c.backend, modelId: c.modelId, reason: why });
|
|
156
|
+
trace.push({ step: agent, agent, capability, backend: c.backend, provider: c.provider, model: c.model, modelId: c.modelId, skipped: true, ok: false, reason: why, startedAt: new Date().toISOString(), durationMs: 0 });
|
|
157
|
+
continue;
|
|
158
|
+
}
|
|
159
|
+
invokedSet.add(c.modelId);
|
|
160
|
+
invokedModels.push(c.modelId);
|
|
161
|
+
const attempt = attempts++;
|
|
162
|
+
const entry = { step: agent, attempt: attempt + 1, agent, capability, backend: c.backend, provider: c.provider, model: c.model, thinking: c.thinking, modelId: c.modelId, degraded: c.degraded, startedAt: new Date().toISOString() };
|
|
163
|
+
const started = Date.now();
|
|
164
|
+
try {
|
|
165
|
+
const result = await fn(c, attempt);
|
|
166
|
+
entry.durationMs = Date.now() - started;
|
|
167
|
+
entry.ok = result.ok;
|
|
168
|
+
entry.reason = result.reason;
|
|
169
|
+
if (result.protocolFailure) entry.protocolFailure = result.protocolFailure; // model-quality failure for task-global marking
|
|
170
|
+
if (result.failureClass) entry.failureClass = result.failureClass;
|
|
171
|
+
// Per-candidate execution telemetry — preserved on the step so a later
|
|
172
|
+
// candidate's record never overwrites an earlier one (e.g. Devin's run is
|
|
173
|
+
// kept even when a subsequent candidate also runs).
|
|
174
|
+
if (result.child) entry.child = result.child;
|
|
175
|
+
if (result.telemetry) entry.telemetry = result.telemetry;
|
|
176
|
+
trace.push(entry);
|
|
177
|
+
if (result.ok) return { ok: true, candidate: c, result, attempts: attempt + 1, escalated: attempt > 0, skipped, invokedModels, counters: { candidatesConsidered: considered, candidatesSkipped: skipped.length, invocationsStarted: invokedModels.length } };
|
|
178
|
+
last = result;
|
|
179
|
+
} catch (e) {
|
|
180
|
+
entry.durationMs = Date.now() - started; entry.ok = false; entry.reason = e.message; trace.push(entry);
|
|
181
|
+
last = { ok: false, reason: e.message };
|
|
182
|
+
}
|
|
183
|
+
onFailure?.(c, String(last.reason ?? 'failed'));
|
|
184
|
+
pack.previous_attempts = pack.previous_attempts ?? [];
|
|
185
|
+
pack.previous_attempts.push({ summary: `${agent} on backend ${c.backend} (${c.modelId}) attempt ${attempt + 1}`, outcome: String(last.reason ?? 'failed').slice(0, 400) });
|
|
186
|
+
}
|
|
187
|
+
const counters = { candidatesConsidered: considered, candidatesSkipped: skipped.length, invocationsStarted: invokedModels.length };
|
|
188
|
+
if (!attempts && skipped.length) {
|
|
189
|
+
return { ok: false, attempts: 0, unavailable: true, exhausted: true, escalationCandidate: null, skipped, invokedModels, counters, last: { ok: false, reason: `all model candidates unavailable: ${skipped.map(s => `${s.modelId} (${s.reason})`).join('; ')}` } };
|
|
190
|
+
}
|
|
191
|
+
const next = eligible[i];
|
|
192
|
+
return { ok: false, attempts, exhausted: !next, escalationCandidate: next ? { backend: next.backend, modelId: next.modelId } : null, last, skipped, invokedModels, counters };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// ---------- full pipeline ----------
|
|
196
|
+
export async function runPipeline({ repoRoot, task, routing, registry, agents, invoke, outDir, maxAttempts = DEFAULT_MAX_ATTEMPTS, skipScout = false, packPath = null, health = null }) {
|
|
197
|
+
const hooks = health ? { skip: health.skip, onFailure: health.report } : {};
|
|
198
|
+
repoRoot = resolve(repoRoot);
|
|
199
|
+
mkdirSync(outDir, { recursive: true });
|
|
200
|
+
const trace = [];
|
|
201
|
+
const byName = Object.fromEntries(agents.map(a => [a.meta.name, a]));
|
|
202
|
+
const scoutAgent = byName.scout, coderAgent = byName.coder;
|
|
203
|
+
if (!scoutAgent || !coderAgent) throw new Error('pipeline requires scout and coder agents');
|
|
204
|
+
const testCommand = detectTestCommand(repoRoot);
|
|
205
|
+
const baseline = runTests(repoRoot, testCommand);
|
|
206
|
+
const summary = { task, repoRoot, testCommand, baselineTestsPass: baseline.ok, steps: trace };
|
|
207
|
+
|
|
208
|
+
// ---- scout -> Context Pack ----
|
|
209
|
+
let pack;
|
|
210
|
+
if (packPath) {
|
|
211
|
+
const n = normalizeContextPack(readFileSync(packPath, 'utf8'), { repoRoot, task, capability: coderAgent.meta.capability });
|
|
212
|
+
if (n.errors.length) throw new Error(`supplied Context Pack invalid: ${n.errors.join('; ')}`);
|
|
213
|
+
pack = n.pack;
|
|
214
|
+
summary.scoutNormalizerReport = n.report;
|
|
215
|
+
summary.packSource = packPath;
|
|
216
|
+
if (testCommand && !(pack.test_commands ?? []).length) pack.test_commands = [testCommand];
|
|
217
|
+
} else if (skipScout) {
|
|
218
|
+
throw new Error('skipScout requires packPath');
|
|
219
|
+
} else {
|
|
220
|
+
const survey = surveyRepo(repoRoot);
|
|
221
|
+
const prompt = scoutPrompt({ task, survey, testOutput: baseline.ok ? null : baseline.output, testCommand });
|
|
222
|
+
const scout = await withEscalation({
|
|
223
|
+
routing, registry, capability: scoutAgent.meta.capability, agent: 'scout', pack: { previous_attempts: [] }, maxAttempts, trace, ...hooks,
|
|
224
|
+
fn: async c => {
|
|
225
|
+
const r = await invoke({ ...c, systemPrompt: scoutAgent.body, prompt, cwd: repoRoot });
|
|
226
|
+
if (!r.ok) return { ok: false, reason: r.error ?? 'invoke failed' };
|
|
227
|
+
writeFileSync(join(outDir, `scout.output.${c.backend}.md`), r.text);
|
|
228
|
+
let n;
|
|
229
|
+
try { n = normalizeContextPack(r.text, { repoRoot, task, capability: coderAgent.meta.capability, producedBy: `scout@${c.modelId}` }); }
|
|
230
|
+
catch (e) { return { ok: false, reason: `scout output not a Context Pack: ${e.message}` }; }
|
|
231
|
+
if (n.errors.length) return { ok: false, reason: `Context Pack invalid: ${n.errors.join('; ')}` };
|
|
232
|
+
if (testCommand && !(n.pack.test_commands ?? []).length) n.pack.test_commands = [testCommand];
|
|
233
|
+
return { ok: true, reason: 'context pack produced', pack: n.pack, report: n.report, raw: r.text };
|
|
234
|
+
},
|
|
235
|
+
});
|
|
236
|
+
if (!scout.ok) { summary.outcome = 'scout-failed'; summary.escalationCandidate = scout.escalationCandidate; writeOut(outDir, summary); return summary; }
|
|
237
|
+
pack = scout.result.pack;
|
|
238
|
+
summary.scoutNormalizerReport = scout.result.report;
|
|
239
|
+
writeFileSync(join(outDir, 'scout.raw.md'), scout.result.raw);
|
|
240
|
+
}
|
|
241
|
+
const packFile = join(outDir, 'context-pack.md');
|
|
242
|
+
writeFileSync(packFile, toMarkdown(pack));
|
|
243
|
+
summary.contextPack = packFile;
|
|
244
|
+
summary.contextPackFiles = pack.relevant_files.map(f => f.path);
|
|
245
|
+
|
|
246
|
+
// ---- coder -> apply -> validate, with escalation ----
|
|
247
|
+
const coder = await withEscalation({
|
|
248
|
+
routing, registry, capability: coderAgent.meta.capability, agent: 'coder', pack, maxAttempts, trace, ...hooks,
|
|
249
|
+
fn: async c => {
|
|
250
|
+
const prompt = coderPrompt({ pack, repoRoot });
|
|
251
|
+
writeFileSync(join(outDir, `coder.input.${c.backend}.md`), `<!-- system -->\n${coderAgent.body}\n\n<!-- user -->\n${prompt}`);
|
|
252
|
+
const r = await invoke({ ...c, systemPrompt: coderAgent.body, prompt, cwd: repoRoot });
|
|
253
|
+
if (!r.ok) return { ok: false, reason: r.error ?? 'invoke failed' };
|
|
254
|
+
writeFileSync(join(outDir, `coder.output.${c.backend}.md`), r.text);
|
|
255
|
+
const blocks = parseFileBlocks(r.text);
|
|
256
|
+
if (!blocks.length) return { ok: false, reason: 'coder returned no FILE blocks' };
|
|
257
|
+
const { applied, rejected } = applyFileBlocks(repoRoot, blocks, pack.relevant_files);
|
|
258
|
+
if (!applied.length) return { ok: false, reason: `all ${rejected.length} file blocks rejected (outside relevant_files)` };
|
|
259
|
+
const tests = runTests(repoRoot, pack.test_commands?.[0] ?? testCommand);
|
|
260
|
+
if (!tests.ok) {
|
|
261
|
+
pack.observed_errors = [tests.output.slice(-2000)];
|
|
262
|
+
return { ok: false, reason: `tests failed after applying ${applied.join(', ')}`, applied, rejected, tests };
|
|
263
|
+
}
|
|
264
|
+
return { ok: true, reason: 'tests pass', applied, rejected, tests };
|
|
265
|
+
},
|
|
266
|
+
});
|
|
267
|
+
summary.outcome = coder.ok ? 'success' : (coder.exhausted ? 'exhausted' : 'escalation-candidate');
|
|
268
|
+
summary.escalated = coder.escalated ?? false;
|
|
269
|
+
summary.attempts = coder.attempts;
|
|
270
|
+
summary.escalationCandidate = coder.escalationCandidate ?? null;
|
|
271
|
+
summary.applied = coder.result?.applied ?? null;
|
|
272
|
+
writeFileSync(packFile, toMarkdown(pack)); // includes previous_attempts / observed_errors updates
|
|
273
|
+
writeOut(outDir, summary);
|
|
274
|
+
return summary;
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
function writeOut(outDir, summary) {
|
|
278
|
+
writeFileSync(join(outDir, 'trace.json'), JSON.stringify(summary, null, 2));
|
|
279
|
+
}
|
package/lib/registry.mjs
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
// Adapter model registry: models.json (shared template) overlaid by models.local.json (gitignored).
|
|
2
|
+
// Validates against routing backends. Contains no provider/model names.
|
|
3
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
4
|
+
|
|
5
|
+
const BINDING_KEYS = new Set(['provider', 'model', 'thinking', 'note', 'vision']);
|
|
6
|
+
const THINKING = new Set(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max']);
|
|
7
|
+
const SECRET_KEYS = /^(apiKey|api_key|token|secret|password|bearer|authorization)$/i;
|
|
8
|
+
const SECRET_VALUE = /(sk-[A-Za-z0-9_-]{20,}|gh[pousr]_[A-Za-z0-9]{20,}|AKIA[0-9A-Z]{16}|-----BEGIN)/;
|
|
9
|
+
|
|
10
|
+
export function validateRegistry(registry, routing, label = 'registry') {
|
|
11
|
+
const errors = [];
|
|
12
|
+
if (!registry || typeof registry !== 'object' || Array.isArray(registry)) return [`${label}: root must be an object`];
|
|
13
|
+
for (const key of Object.keys(registry)) {
|
|
14
|
+
if (!['version', 'backends', '$comment'].includes(key)) errors.push(`${label}: unknown top-level key "${key}"`);
|
|
15
|
+
}
|
|
16
|
+
if (registry.version !== 1) errors.push(`${label}: version must be 1`);
|
|
17
|
+
if (!registry.backends || typeof registry.backends !== 'object' || Array.isArray(registry.backends)) {
|
|
18
|
+
errors.push(`${label}: backends must be an object`);
|
|
19
|
+
return errors;
|
|
20
|
+
}
|
|
21
|
+
for (const [backend, binding] of Object.entries(registry.backends)) {
|
|
22
|
+
if (routing && !(backend in routing.backends)) errors.push(`${label}: backend "${backend}" is not defined in routing`);
|
|
23
|
+
if (!binding || typeof binding !== 'object' || Array.isArray(binding)) { errors.push(`${label}: backend "${backend}" binding must be an object`); continue; }
|
|
24
|
+
for (const key of Object.keys(binding)) {
|
|
25
|
+
if (SECRET_KEYS.test(key)) errors.push(`${label}: backend "${backend}" must not store credentials ("${key}")`);
|
|
26
|
+
else if (!BINDING_KEYS.has(key)) errors.push(`${label}: backend "${backend}" has unknown key "${key}"`);
|
|
27
|
+
if (typeof binding[key] === 'string' && SECRET_VALUE.test(binding[key])) errors.push(`${label}: backend "${backend}".${key} looks like a credential`);
|
|
28
|
+
}
|
|
29
|
+
if (binding.provider !== undefined && (typeof binding.provider !== 'string' || !binding.provider.trim())) errors.push(`${label}: backend "${backend}".provider must be a non-empty string`);
|
|
30
|
+
if (binding.model !== undefined && (typeof binding.model !== 'string' || !binding.model.trim())) errors.push(`${label}: backend "${backend}".model must be a non-empty string`);
|
|
31
|
+
if (binding.thinking !== undefined && !THINKING.has(binding.thinking)) errors.push(`${label}: backend "${backend}".thinking must be one of ${[...THINKING].join('|')}`);
|
|
32
|
+
}
|
|
33
|
+
return errors;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Shallow-per-backend merge: local binding fields override template fields; local may add backends. */
|
|
37
|
+
export function mergeRegistries(base, local) {
|
|
38
|
+
const out = { version: 1, backends: {} };
|
|
39
|
+
for (const [b, v] of Object.entries(base?.backends ?? {})) out.backends[b] = { ...v };
|
|
40
|
+
for (const [b, v] of Object.entries(local?.backends ?? {})) out.backends[b] = { ...(out.backends[b] ?? {}), ...v };
|
|
41
|
+
return out;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function isPlaceholder(binding) {
|
|
45
|
+
return !binding || !binding.provider || !binding.model || /^TODO/i.test(binding.provider) || /^TODO/i.test(binding.model);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Load <adapterDir>/models.json overlaid with <adapterDir>/models.local.json (if present).
|
|
50
|
+
* Throws on validation errors of either file or of the merged result.
|
|
51
|
+
*/
|
|
52
|
+
export function loadRegistry(modelsPath, localPath, routing) {
|
|
53
|
+
const base = JSON.parse(readFileSync(modelsPath, 'utf8'));
|
|
54
|
+
const errors = validateRegistry(base, routing, 'models.json');
|
|
55
|
+
let local = null;
|
|
56
|
+
if (localPath && existsSync(localPath)) {
|
|
57
|
+
local = JSON.parse(readFileSync(localPath, 'utf8'));
|
|
58
|
+
errors.push(...validateRegistry(local, routing, 'models.local.json'));
|
|
59
|
+
}
|
|
60
|
+
if (errors.length) throw new Error(errors.join('\n'));
|
|
61
|
+
const merged = mergeRegistries(base, local);
|
|
62
|
+
return { registry: merged, sources: { models: modelsPath, local: local ? localPath : null } };
|
|
63
|
+
}
|
package/lib/resolve.mjs
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
// agent -> capability -> routing backends -> registry binding -> concrete model candidates.
|
|
2
|
+
// Generic over adapters: the registry decides what "provider/model" means. No names hardcoded.
|
|
3
|
+
import { resolveBackends, degradedBackends } from './routing.mjs';
|
|
4
|
+
import { isPlaceholder } from './registry.mjs';
|
|
5
|
+
|
|
6
|
+
/** Format a binding as pi-style "provider/model[:thinking]" (pi CLI --model / pi-subagents model syntax). */
|
|
7
|
+
export function formatModelId(binding, { withThinking = true } = {}) {
|
|
8
|
+
const id = `${binding.provider}/${binding.model}`;
|
|
9
|
+
return withThinking && binding.thinking ? `${id}:${binding.thinking}` : id;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/** Ordered candidate chain for one capability. Placeholders/unbound backends are excluded but reported. */
|
|
13
|
+
export function resolveCapability(routing, registry, capability) {
|
|
14
|
+
const chain = resolveBackends(routing, capability);
|
|
15
|
+
const degraded = new Set(degradedBackends(routing, capability));
|
|
16
|
+
const candidates = [], unbound = [], placeholder = [];
|
|
17
|
+
for (const backend of chain) {
|
|
18
|
+
const binding = registry.backends?.[backend];
|
|
19
|
+
if (!binding) { unbound.push(backend); continue; }
|
|
20
|
+
if (isPlaceholder(binding)) { placeholder.push(backend); continue; }
|
|
21
|
+
candidates.push({ backend, provider: binding.provider, model: binding.model, thinking: binding.thinking, modelId: formatModelId(binding), degraded: degraded.has(backend) || undefined });
|
|
22
|
+
}
|
|
23
|
+
return { capability, chain, candidates, unbound, placeholder };
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** Resolve every agent (from lib/agents.mjs loadAgents) to its capability chain. */
|
|
27
|
+
export function resolveAgents(agents, routing, registry) {
|
|
28
|
+
const out = {};
|
|
29
|
+
for (const agent of agents) {
|
|
30
|
+
const capability = agent.meta.capability;
|
|
31
|
+
if (!(capability in routing.capabilities)) continue;
|
|
32
|
+
out[agent.meta.name] = { agent: agent.meta.name, ...resolveCapability(routing, registry, capability) };
|
|
33
|
+
}
|
|
34
|
+
return out;
|
|
35
|
+
}
|
package/lib/routing.mjs
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// Minimal routing config loader/validator. No provider or model names are hardcoded here;
|
|
2
|
+
// everything comes from routing.json (logical backends) and adapter model maps (concrete bindings).
|
|
3
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
4
|
+
import { join, dirname } from 'node:path';
|
|
5
|
+
|
|
6
|
+
const NAME = /^[a-z][a-z0-9-]*$/;
|
|
7
|
+
const TIERS = new Set(['free', 'low', 'mid', 'high']);
|
|
8
|
+
|
|
9
|
+
export function validateRouting(config) {
|
|
10
|
+
const errors = [];
|
|
11
|
+
const err = m => errors.push(m);
|
|
12
|
+
if (!config || typeof config !== 'object' || Array.isArray(config)) return ['routing: root must be an object'];
|
|
13
|
+
if (config.version !== 1) err('routing: version must be 1');
|
|
14
|
+
const backends = config.backends;
|
|
15
|
+
if (!backends || typeof backends !== 'object' || Object.keys(backends).length === 0) {
|
|
16
|
+
err('routing: backends must be a non-empty object');
|
|
17
|
+
} else {
|
|
18
|
+
for (const [name, b] of Object.entries(backends)) {
|
|
19
|
+
if (!NAME.test(name)) err(`routing: invalid backend name "${name}"`);
|
|
20
|
+
if (!b || typeof b !== 'object') { err(`routing: backend "${name}" must be an object`); continue; }
|
|
21
|
+
if (b.tier !== undefined && !TIERS.has(b.tier)) err(`routing: backend "${name}" has invalid tier "${b.tier}"`);
|
|
22
|
+
if (b.vision !== undefined && typeof b.vision !== 'boolean') err(`routing: backend "${name}".vision must be boolean`);
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
const caps = config.capabilities;
|
|
26
|
+
if (!caps || typeof caps !== 'object' || Object.keys(caps).length === 0) {
|
|
27
|
+
err('routing: capabilities must be a non-empty object');
|
|
28
|
+
} else {
|
|
29
|
+
for (const [name, c] of Object.entries(caps)) {
|
|
30
|
+
if (!NAME.test(name)) err(`routing: invalid capability name "${name}"`);
|
|
31
|
+
if (!c || typeof c !== 'object') { err(`routing: capability "${name}" must be an object`); continue; }
|
|
32
|
+
if (typeof c.primary !== 'string') err(`routing: capability "${name}" requires string primary`);
|
|
33
|
+
else if (backends && !(c.primary in backends)) err(`routing: capability "${name}" primary "${c.primary}" is not a defined backend`);
|
|
34
|
+
const fallback = c.fallback ?? [];
|
|
35
|
+
if (!Array.isArray(fallback)) err(`routing: capability "${name}".fallback must be an array`);
|
|
36
|
+
else {
|
|
37
|
+
const seen = new Set();
|
|
38
|
+
for (const f of fallback) {
|
|
39
|
+
if (typeof f !== 'string') { err(`routing: capability "${name}" fallback entries must be strings`); continue; }
|
|
40
|
+
if (backends && !(f in backends)) err(`routing: capability "${name}" fallback "${f}" is not a defined backend`);
|
|
41
|
+
if (f === c.primary) err(`routing: capability "${name}" fallback repeats primary "${f}"`);
|
|
42
|
+
if (seen.has(f)) err(`routing: capability "${name}" fallback repeats "${f}"`);
|
|
43
|
+
seen.add(f);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
// requires.vision is enforced on primary; fallbacks may be degraded (reported by resolveBackends).
|
|
47
|
+
if (c.requires?.vision === true && backends && backends[c.primary] && backends[c.primary].vision !== true) {
|
|
48
|
+
err(`routing: capability "${name}" requires vision but primary backend "${c.primary}" is not vision-capable`);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
if (config.escalation?.ladders) {
|
|
53
|
+
for (const [ladder, steps] of Object.entries(config.escalation.ladders)) {
|
|
54
|
+
if (!Array.isArray(steps) || steps.length === 0) { err(`routing: ladder "${ladder}" must be a non-empty array`); continue; }
|
|
55
|
+
for (const s of steps) if (caps && !(s in caps)) err(`routing: ladder "${ladder}" references unknown capability "${s}"`);
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
return errors;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// Machine-local capability overrides: null disables a shared capability;
|
|
62
|
+
// an object overrides its route or adds a new capability. The shared file is
|
|
63
|
+
// never modified. Deleted capabilities are removed from escalation ladders
|
|
64
|
+
// in the effective view; agents referencing them are unavailable.
|
|
65
|
+
export function mergeLocalRouting(base, local) {
|
|
66
|
+
if (!local || local.version !== 1 || !local.capabilities || typeof local.capabilities !== 'object' || Array.isArray(local.capabilities) ||
|
|
67
|
+
Object.keys(local).some(k => !['version', 'capabilities', '$comment'].includes(k))) {
|
|
68
|
+
throw new Error('routing.local.json: expected version 1 and capabilities object');
|
|
69
|
+
}
|
|
70
|
+
const config = { ...base, capabilities: { ...base.capabilities } };
|
|
71
|
+
for (const [name, override] of Object.entries(local.capabilities)) {
|
|
72
|
+
if (!NAME.test(name)) throw new Error(`routing.local.json: invalid capability name "${name}"`);
|
|
73
|
+
if (override === null) {
|
|
74
|
+
if (!(name in base.capabilities)) throw new Error(`routing.local.json: cannot delete unknown capability "${name}"`);
|
|
75
|
+
delete config.capabilities[name];
|
|
76
|
+
continue;
|
|
77
|
+
}
|
|
78
|
+
if (!override || typeof override !== 'object' || Array.isArray(override) ||
|
|
79
|
+
Object.keys(override).some(k => !['primary', 'fallback', 'description'].includes(k))) {
|
|
80
|
+
throw new Error(`routing.local.json: invalid capability override "${name}"`);
|
|
81
|
+
}
|
|
82
|
+
if (override.primary !== undefined && typeof override.primary !== 'string') throw new Error(`routing.local.json: capability "${name}" primary must be a string`);
|
|
83
|
+
if (override.fallback !== undefined && (!Array.isArray(override.fallback) || override.fallback.some(value => typeof value !== 'string'))) throw new Error(`routing.local.json: capability "${name}" fallback must be an array of strings`);
|
|
84
|
+
if (override.description !== undefined && typeof override.description !== 'string') throw new Error(`routing.local.json: capability "${name}" description must be a string`);
|
|
85
|
+
config.capabilities[name] = { ...(base.capabilities[name] ?? {}), ...override };
|
|
86
|
+
}
|
|
87
|
+
if (base.escalation?.ladders) {
|
|
88
|
+
const ladders = Object.fromEntries(Object.entries(base.escalation.ladders)
|
|
89
|
+
.map(([name, steps]) => [name, steps.filter(step => step in config.capabilities)])
|
|
90
|
+
.filter(([, steps]) => steps.length));
|
|
91
|
+
config.escalation = { ...base.escalation, ladders };
|
|
92
|
+
}
|
|
93
|
+
const errors = validateRouting(config);
|
|
94
|
+
if (errors.length) throw new Error(errors.join('\n'));
|
|
95
|
+
return config;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export function loadRouting(path) {
|
|
99
|
+
const config = JSON.parse(readFileSync(path, 'utf8'));
|
|
100
|
+
const errors = validateRouting(config);
|
|
101
|
+
if (errors.length) throw new Error(errors.join('\n'));
|
|
102
|
+
const localPath = join(dirname(path), 'routing.local.json');
|
|
103
|
+
return existsSync(localPath) ? mergeLocalRouting(config, JSON.parse(readFileSync(localPath, 'utf8'))) : config;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Ordered backend candidates for a capability: primary first, then fallbacks. */
|
|
107
|
+
export function resolveBackends(config, capability) {
|
|
108
|
+
const cap = config.capabilities[capability];
|
|
109
|
+
if (!cap) throw new Error(`routing: unknown capability "${capability}"`);
|
|
110
|
+
return [cap.primary, ...(cap.fallback ?? [])];
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Backends in the chain that do not satisfy the capability's `requires` (e.g. non-vision fallback). */
|
|
114
|
+
export function degradedBackends(config, capability) {
|
|
115
|
+
const cap = config.capabilities[capability];
|
|
116
|
+
if (!cap?.requires) return [];
|
|
117
|
+
return resolveBackends(config, capability).filter(b => {
|
|
118
|
+
const backend = config.backends[b] ?? {};
|
|
119
|
+
return cap.requires.vision === true && backend.vision !== true;
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Resolve a capability to concrete provider/model candidates using an adapter model map
|
|
125
|
+
* of shape { backends: { <backend>: { provider, model, ... } } }. Backends without a
|
|
126
|
+
* binding are skipped (reported in `unbound`), so a missing local model never breaks routing.
|
|
127
|
+
*/
|
|
128
|
+
export function resolveModels(config, modelMap, capability) {
|
|
129
|
+
const bound = [], unbound = [];
|
|
130
|
+
const degraded = degradedBackends(config, capability);
|
|
131
|
+
for (const backend of resolveBackends(config, capability)) {
|
|
132
|
+
const binding = modelMap?.backends?.[backend];
|
|
133
|
+
if (binding && binding.provider && binding.model) bound.push({ backend, ...binding, degraded: degraded.includes(backend) || undefined });
|
|
134
|
+
else unbound.push(backend);
|
|
135
|
+
}
|
|
136
|
+
return { capability, candidates: bound, unbound, degraded };
|
|
137
|
+
}
|