@hecer/yoke 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +18 -3
  4. package/README.md +87 -21
  5. package/bench/README.md +46 -14
  6. package/bench/RESULTS.md +108 -11
  7. package/bench/analyze-routing-study.mjs +96 -0
  8. package/bench/fixtures/routing-queue/.yoke/config.yaml +30 -0
  9. package/bench/fixtures/routing-queue/.yoke/prd.yaml +24 -0
  10. package/bench/fixtures/routing-queue/bench-verify.mjs +9 -0
  11. package/bench/fixtures/routing-queue/package.json +9 -0
  12. package/bench/fixtures/routing-queue/src/task-queue.mjs +1 -0
  13. package/bench/fixtures/routing-queue/tests/STORY-1.test.mjs +68 -0
  14. package/bench/fixtures/routing-queue/tests/STORY-2.test.mjs +72 -0
  15. package/bench/results/routing-queue-codex-routing-off-2026-08-01T21-44-00.json +46 -0
  16. package/bench/results/routing-queue-codex-routing-on-2026-08-01T21-50-31.json +109 -0
  17. package/bench/results/yoke-codex-study-codex-routing-off-2026-08-02T07-58-56.json +65 -0
  18. package/bench/results/yoke-codex-study-codex-routing-off-2026-08-02T08-08-48.json +65 -0
  19. package/bench/results/yoke-codex-study-codex-routing-off-2026-08-02T08-34-45.json +65 -0
  20. package/bench/results/yoke-codex-study-codex-routing-on-2026-08-02T07-51-59.json +168 -0
  21. package/bench/results/yoke-codex-study-codex-routing-on-2026-08-02T08-18-59.json +168 -0
  22. package/bench/results/yoke-codex-study-codex-routing-on-2026-08-02T08-25-25.json +168 -0
  23. package/bench/results/yoke-large-codex-routing-off-2026-08-01T22-58-01.json +54 -0
  24. package/bench/results/yoke-large-codex-routing-on-2026-08-01T23-09-39.json +133 -0
  25. package/bench/run-large.mjs +188 -0
  26. package/bench/run.mjs +42 -21
  27. package/dist/agents/providers.js +21 -4
  28. package/dist/agents/telemetry.js +46 -7
  29. package/dist/cli.js +7 -5
  30. package/dist/loop/decision.js +3 -1
  31. package/dist/loop/loop.js +10 -0
  32. package/dist/loop/reporter.js +9 -0
  33. package/dist/loop/run-command.js +35 -7
  34. package/dist/loop/runner.js +68 -9
  35. package/dist/retrofit/config.js +21 -0
  36. package/dist/review/command.js +2 -2
  37. package/dist/routing/registry.js +92 -0
  38. package/dist/routing/router.js +195 -0
  39. package/dist/setup/command.js +23 -1
  40. package/gemini-extension.json +1 -1
  41. package/package.json +6 -4
@@ -13,6 +13,7 @@ import { resolveCommitIdentity } from './identity.js';
13
13
  import { runAudit } from '../audit/command.js';
14
14
  import { detectHostAgent, resolveRunnerAgent } from '../agents/host.js';
15
15
  import { clearDecisionResume, decisionProcessingExists, decisionRequestId, formatPendingDecision, readPendingDecision, writeDecisionResume, } from './decision.js';
16
+ import { makeAdaptiveRunner } from '../routing/router.js';
16
17
  export const DEFAULT_IDLE_MINUTES = 20;
17
18
  const STALE_MINUTES = 20; // a running status older than this likely means the loop died
18
19
  export function relativeTime(fromIso, now) {
@@ -142,6 +143,11 @@ export function runLoopCommand(targetDir, opts) {
142
143
  }
143
144
  const idleMs = resolveIdleMs(opts.timeoutMinutes, config.loop.timeoutMinutes);
144
145
  const permissions = opts.permissions ?? config.runner?.permissions ?? 'safe';
146
+ const routingEnabled = opts.routing ?? config.routing?.enabled ?? false;
147
+ if (routingEnabled && (!config.routing || config.routing.workers.length === 0)) {
148
+ console.error('Adaptive routing was requested, but no worker profiles are configured. Run yoke setup . --routing or add routing.workers to .yoke/config.yaml.');
149
+ return 2;
150
+ }
145
151
  let runner = opts.runner;
146
152
  if (!runner) {
147
153
  if (!available(runnerAgent)) {
@@ -150,14 +156,34 @@ export function runLoopCommand(targetDir, opts) {
150
156
  }
151
157
  // Token reporting is part of the machine interface: in --json mode a claude
152
158
  // runner switches to stream-json so cumulative usage rides on every status.
153
- runner = makeRunner(runnerAgent, idleMs, {
159
+ const runnerOpts = {
154
160
  tokenReport: opts.json === true,
155
161
  onAmbiguity: opts.decisionPolicy ?? opts.onAmbiguity ?? config.loop.decisionPolicy ?? config.loop.onAmbiguity ?? 'auto',
156
162
  perfCommand: config.perf?.command,
157
163
  permissions,
158
- });
164
+ selection: {
165
+ model: config.runner?.model,
166
+ reasoningEffort: config.runner?.reasoningEffort,
167
+ bare: config.runner?.bare,
168
+ ...((routingEnabled || opts.routing === false) ? { nativeMultiAgent: false } : {}),
169
+ },
170
+ };
171
+ runner = routingEnabled && config.routing
172
+ ? makeAdaptiveRunner({
173
+ parent: runnerAgent,
174
+ parentSelection: runnerOpts.selection,
175
+ orchestratorSelection: config.routing.orchestrator ?? runnerOpts.selection,
176
+ workers: config.routing.workers,
177
+ strategy: config.routing.strategy,
178
+ maxCandidates: config.routing.maxCandidates,
179
+ idleTimeoutMs: idleMs,
180
+ permissions,
181
+ runnerOpts,
182
+ isAvailable: available,
183
+ })
184
+ : makeRunner(runnerAgent, idleMs, runnerOpts);
159
185
  const announce = opts.json ? console.error : console.log;
160
- announce(`Runner: ${runnerAgent} · permissions: ${permissions} · cwd: ${targetDir}`);
186
+ announce(`Runner: ${runnerAgent} · permissions: ${permissions} · routing: ${routingEnabled ? 'on' : 'off'} · cwd: ${targetDir}`);
161
187
  }
162
188
  let review = opts.reviewRunner;
163
189
  if (!review && (opts.review || opts.reviewer)) {
@@ -196,6 +222,7 @@ export function runLoopCommand(targetDir, opts) {
196
222
  }
197
223
  try {
198
224
  const reporter = opts.reporter ?? makeReporter(targetDir, { json: opts.json });
225
+ const maxIterations = opts.maxIterations ?? Number.POSITIVE_INFINITY;
199
226
  const result = runLoop({
200
227
  prdPath: path,
201
228
  targetDir,
@@ -205,7 +232,7 @@ export function runLoopCommand(targetDir, opts) {
205
232
  verify,
206
233
  perf,
207
234
  audit,
208
- maxIterations: opts.maxIterations,
235
+ maxIterations,
209
236
  isolate: (opts.parallel ?? 1) > 1 ? true : (opts.isolate ?? false),
210
237
  review,
211
238
  reporter,
@@ -218,7 +245,7 @@ export function runLoopCommand(targetDir, opts) {
218
245
  storyId: pendingDecision.storyId,
219
246
  requestId: decisionRequestId(pendingDecision),
220
247
  answered: false,
221
- maxIterations: opts.maxIterations,
248
+ ...(opts.maxIterations !== undefined ? { maxIterations: opts.maxIterations } : {}),
222
249
  agent: runnerAgent,
223
250
  isolate: opts.isolate ?? false,
224
251
  reviewer: opts.reviewer,
@@ -237,6 +264,7 @@ export function runLoopCommand(targetDir, opts) {
237
264
  : config.loop.decisionPolicy),
238
265
  permissions,
239
266
  parallel: opts.parallel ?? 1,
267
+ routing: routingEnabled,
240
268
  });
241
269
  }
242
270
  else
@@ -252,8 +280,8 @@ export function runLoopCommand(targetDir, opts) {
252
280
  say(`Loop ${result.status} after ${result.iterations} iteration(s): ${result.finalProgress.passed}/${result.finalProgress.total} stories pass`);
253
281
  if (result.reason)
254
282
  say(`Reason: ${result.reason}`);
255
- if (result.reason && /api key|please run \/login|not logged in/i.test(result.reason)) {
256
- say('Hint: the agent CLI has no credentials in this environment. Set ANTHROPIC_API_KEY or log the agent in for headless use.');
283
+ if (result.reason && /api key|please run \/login|not logged in|auth/i.test(result.reason)) {
284
+ say('Hint: the agent CLI has no credentials in this environment. Set ANTHROPIC_API_KEY, GEMINI_API_KEY, or OPENAI_API_KEY, or log the agent in for headless use.');
257
285
  }
258
286
  // Exit codes: 0 complete · 1 blocked/cap-reached · 2 config error (handled above) · 3 paused (loop.pause consumed at a story boundary)
259
287
  if (result.status === 'complete')
@@ -40,7 +40,9 @@ export function buildReviewPrompt(story, context, verdictPath) {
40
40
  ];
41
41
  if (context)
42
42
  lines.push('', context);
43
- lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
43
+ lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', verdictPath
44
+ ? 'Do not modify project source, tests, configuration, or generated artifacts. The verdict file named below is the only permitted write. Do not commit.'
45
+ : 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
44
46
  if (verdictPath)
45
47
  lines.push('', formatReviewContract(verdictPath));
46
48
  return lines.join('\n');
@@ -53,7 +55,9 @@ export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
53
55
  ];
54
56
  if (focus)
55
57
  lines.push(`Pay particular attention to: ${focus}.`);
56
- lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
58
+ lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', verdictPath
59
+ ? 'Do not modify project source, tests, configuration, or generated artifacts. The verdict file named below is the only permitted write. Do not commit.'
60
+ : 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
57
61
  if (verdictPath)
58
62
  lines.push('', formatReviewContract(verdictPath));
59
63
  return lines.join('\n');
@@ -64,8 +68,8 @@ export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
64
68
  // and falsely marks the story done. Granting autonomous permissions makes the
65
69
  // implementer actually able to write files and run the verify command.
66
70
  // (The loop is opt-in and scoped to the target project dir.)
67
- export function agentInvocation(agent, prompt, cwd, permissions = 'safe') {
68
- return buildProviderInvocation(agent, prompt, cwd, permissions);
71
+ export function agentInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}) {
72
+ return buildProviderInvocation(agent, prompt, cwd, permissions, selection);
69
73
  }
70
74
  export function claudeInvocation(prompt, cwd) {
71
75
  return agentInvocation('claude', prompt, cwd);
@@ -82,8 +86,8 @@ export function claudeStreamJsonInvocation(prompt, cwd) {
82
86
  // the user saw dead air the whole time. stream-json emits per-message output,
83
87
  // which doubles as liveness. Token usage rides along for free. Other agents
84
88
  // keep their plain invocation (no machine-readable stream to gain).
85
- export function runnerInvocation(agent, prompt, cwd, _tokenReport = false, permissions = 'safe') {
86
- return buildProviderInvocation(agent, prompt, cwd, permissions);
89
+ export function runnerInvocation(agent, prompt, cwd, _tokenReport = false, permissions = 'safe', selection = {}) {
90
+ return buildProviderInvocation(agent, prompt, cwd, permissions, selection);
87
91
  }
88
92
  // Parse claude stream-json output into cumulative token usage. Defensive by design:
89
93
  // non-JSON lines and unknown message shapes are ignored. The final "result" message
@@ -200,6 +204,51 @@ function runCliCapture(inv) {
200
204
  ? execSync(win32CommandString(inv.command, inv.args), opts)
201
205
  : execFileSync(inv.command, inv.args, opts);
202
206
  }
207
+ // Reviews have a machine-readable result file, so their console stream is not
208
+ // the result channel. Buffer stderr to preserve the provider's actual failure
209
+ // (authentication, sandbox startup, quota, etc.) in loop-status.json instead of
210
+ // reducing every failure to Node's generic "Command failed" message. The inner
211
+ // watchdog still observes child output live and enforces the idle timeout.
212
+ function runReviewCli(inv) {
213
+ const opts = {
214
+ cwd: inv.cwd,
215
+ input: inv.input,
216
+ stdio: ['pipe', 'pipe', 'pipe'],
217
+ encoding: 'utf8',
218
+ maxBuffer: 64 * 1024 * 1024,
219
+ };
220
+ if (process.platform === 'win32')
221
+ execSync(win32CommandString(inv.command, inv.args), opts);
222
+ else
223
+ execFileSync(inv.command, inv.args, opts);
224
+ }
225
+ function processFailureSummary(error) {
226
+ const message = error instanceof Error ? error.message : String(error);
227
+ const value = error?.stderr;
228
+ const stderr = Buffer.isBuffer(value) ? value.toString('utf8') : typeof value === 'string' ? value : '';
229
+ const clean = stderr.replace(/\x1B\[[0-?]*[ -/]*[@-~]/gu, '').trim();
230
+ if (!clean)
231
+ return message;
232
+ const tail = clean.length > 4_000 ? `…${clean.slice(-4_000)}` : clean;
233
+ return `${message}; stderr: ${tail}`;
234
+ }
235
+ /** Run a provider invocation with stdout captured for structured control-plane calls. */
236
+ export function runCapturedAgent(agent, inv) {
237
+ try {
238
+ const output = runCliCapture(inv);
239
+ return { success: true, output, summary: 'exited 0', tokens: parseProviderTelemetry(agent, output.split(/\r?\n/)).tokens };
240
+ }
241
+ catch (error) {
242
+ const partial = error.stdout;
243
+ const output = partial == null ? '' : String(partial);
244
+ return {
245
+ success: false,
246
+ output,
247
+ summary: error.message,
248
+ tokens: output ? parseProviderTelemetry(agent, output.split(/\r?\n/)).tokens : undefined,
249
+ };
250
+ }
251
+ }
203
252
  // Probe whether a CLI is on PATH via `<command> --version`. Same win32/other split
204
253
  // as runCli to stay DEP0190-free. Never throws. Timeout is generous because some
205
254
  // agent CLIs cold-start slowly (gemini needs ~6s on Windows; 5s misreported it
@@ -229,13 +278,23 @@ export function runAgent(inv) {
229
278
  return { success: false, summary: e.message };
230
279
  }
231
280
  }
281
+ /** Run a reviewer while retaining bounded stderr diagnostics on failure. */
282
+ export function runReviewAgent(inv) {
283
+ try {
284
+ runReviewCli(inv);
285
+ return { success: true, summary: 'exited 0' };
286
+ }
287
+ catch (error) {
288
+ return { success: false, summary: processFailureSummary(error) };
289
+ }
290
+ }
232
291
  export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
233
292
  // Claude always streams (see runnerInvocation) — capture the stream so tokens are
234
293
  // always reported; other agents keep inherit stdio. opts.tokenReport is now
235
294
  // redundant for claude and meaningless elsewhere; kept for caller compatibility.
236
295
  const captureTokens = true;
237
296
  return (ctx) => {
238
- const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens, opts.permissions ?? 'safe');
297
+ const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens, opts.permissions ?? 'safe', opts.selection);
239
298
  const inv = buildWatchdogInvocation(base, idleTimeoutMs);
240
299
  if (captureTokens) {
241
300
  const capture = opts.execCapture ?? runCliCapture;
@@ -264,7 +323,7 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
264
323
  };
265
324
  }
266
325
  export const claudeRunner = makeRunner('claude');
267
- export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runCli) {
326
+ export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runReviewCli) {
268
327
  return (ctx) => {
269
328
  const verdictPath = reviewVerdictPath(ctx.targetDir);
270
329
  mkdirSync(join(ctx.targetDir, '.yoke'), { recursive: true });
@@ -276,7 +335,7 @@ export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runCli) {
276
335
  exec(inv);
277
336
  }
278
337
  catch (e) {
279
- processFailure = e.message;
338
+ processFailure = processFailureSummary(e);
280
339
  }
281
340
  try {
282
341
  const verdict = readReviewVerdict(verdictPath);
@@ -6,6 +6,14 @@ const AgentSchema = z.enum(['claude', 'codex', 'gemini']);
6
6
  const CodeGraphSchema = z.enum(['graphify', 'serena']);
7
7
  const SmokeFlowSchema = z.object({ name: z.string().min(1), path: z.string().min(1), landmark: z.string().optional() });
8
8
  const SmokeSchema = z.object({ baseUrl: z.string().min(1), flows: z.array(SmokeFlowSchema).min(1) });
9
+ const RoutingWorkerSchema = z.object({
10
+ id: z.string().regex(/^[a-z0-9][a-z0-9-]*$/),
11
+ agent: AgentSchema,
12
+ model: z.string().min(1).optional(),
13
+ reasoningEffort: z.string().min(1).optional(),
14
+ costTier: z.enum(['low', 'medium', 'high']).default('medium'),
15
+ capabilities: z.array(z.string().min(1)).default([]),
16
+ });
9
17
  export const YokeConfigSchema = z.object({
10
18
  canonVersion: z.string().min(1),
11
19
  agents: z.array(AgentSchema),
@@ -19,8 +27,21 @@ export const YokeConfigSchema = z.object({
19
27
  }),
20
28
  runner: z.object({
21
29
  agent: AgentSchema.optional(),
30
+ model: z.string().min(1).optional(),
31
+ reasoningEffort: z.string().min(1).optional(),
32
+ bare: z.boolean().optional(),
22
33
  permissions: z.enum(['safe', 'unsafe', 'read-only']).optional(),
23
34
  }).optional(),
35
+ routing: z.object({
36
+ enabled: z.boolean(),
37
+ strategy: z.enum(['balanced', 'cost', 'speed', 'quality']).default('balanced'),
38
+ maxCandidates: z.number().int().min(1).max(5).default(3),
39
+ orchestrator: z.object({
40
+ model: z.string().min(1).optional(),
41
+ reasoningEffort: z.string().min(1).optional(),
42
+ }).optional(),
43
+ workers: z.array(RoutingWorkerSchema).max(12).default([]),
44
+ }).optional(),
24
45
  commit: z.object({
25
46
  authorName: z.string().min(1).optional(),
26
47
  authorEmail: z.string().email().optional(),
@@ -1,4 +1,4 @@
1
- import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runAgent, isAgentAvailable, } from '../loop/runner.js';
1
+ import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runReviewAgent, isAgentAvailable, } from '../loop/runner.js';
2
2
  import { resolveIdleMs } from '../loop/run-command.js';
3
3
  import { existsSync, mkdirSync, rmSync, rmdirSync } from 'node:fs';
4
4
  import { join } from 'node:path';
@@ -46,7 +46,7 @@ export function runReview(targetDir, opts = {}) {
46
46
  const inv = agentInvocation(reviewer, prompt, targetDir, 'safe');
47
47
  const say = opts.json ? console.error : console.log;
48
48
  say(`Reviewing ${scope} with ${reviewer}...`);
49
- const run = opts.run ?? ((i) => runAgent(buildWatchdogInvocation(i, idleMs)));
49
+ const run = opts.run ?? ((i) => runReviewAgent(buildWatchdogInvocation(i, idleMs)));
50
50
  const processResult = run(inv);
51
51
  let verdict;
52
52
  try {
@@ -0,0 +1,92 @@
1
+ import { createHash, randomUUID } from 'node:crypto';
2
+ import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from 'node:fs';
3
+ import { homedir } from 'node:os';
4
+ import { join, resolve } from 'node:path';
5
+ const HISTORY_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000;
6
+ export function registryDir() {
7
+ if (process.env.YOKE_REGISTRY_DIR)
8
+ return resolve(process.env.YOKE_REGISTRY_DIR);
9
+ const base = process.env.LOCALAPPDATA || process.env.XDG_STATE_HOME || join(homedir(), '.yoke');
10
+ return join(base, 'Yoke', 'routing-registry');
11
+ }
12
+ function eventsDir() {
13
+ return join(registryDir(), 'events');
14
+ }
15
+ export function projectHash(targetDir) {
16
+ return createHash('sha256').update(`${process.platform}\0${resolve(targetDir)}`).digest('hex').slice(0, 16);
17
+ }
18
+ export function storyHash(project, storyId) {
19
+ return createHash('sha256').update(`${project}\0${storyId}`).digest('hex').slice(0, 16);
20
+ }
21
+ /**
22
+ * Immutable one-file-per-event writes avoid a shared mutable JSON file. Concurrent
23
+ * Yoke processes never hold a registry lock and cannot overwrite each other.
24
+ */
25
+ export function recordRoutingObservation(observation) {
26
+ try {
27
+ const dir = eventsDir();
28
+ mkdirSync(dir, { recursive: true });
29
+ const event = {
30
+ schemaVersion: 1,
31
+ eventId: randomUUID(),
32
+ recordedAt: new Date().toISOString(),
33
+ ...observation,
34
+ };
35
+ const stamp = event.recordedAt.replace(/[:.]/g, '-');
36
+ writeFileSync(join(dir, `${stamp}-${process.pid}-${event.eventId}.json`), JSON.stringify(event), { flag: 'wx' });
37
+ return event;
38
+ }
39
+ catch {
40
+ // Routing evidence is observability. A registry failure must never block a story.
41
+ return null;
42
+ }
43
+ }
44
+ export function readRoutingObservations(limit = 1000) {
45
+ const dir = eventsDir();
46
+ if (!existsSync(dir))
47
+ return [];
48
+ const files = readdirSync(dir).filter(file => file.endsWith('.json')).sort().slice(-limit);
49
+ const observations = [];
50
+ for (const file of files) {
51
+ try {
52
+ const value = JSON.parse(readFileSync(join(dir, file), 'utf8'));
53
+ if (value.schemaVersion === 1 && typeof value.selected === 'string')
54
+ observations.push(value);
55
+ }
56
+ catch { /* ignore partial/corrupt evidence without affecting routing */ }
57
+ }
58
+ return observations;
59
+ }
60
+ export function historyForWorkers(workers) {
61
+ const wanted = new Map(workers.map(worker => [worker.id, worker]));
62
+ const grouped = new Map();
63
+ const oldestUseful = Date.now() - HISTORY_MAX_AGE_MS;
64
+ for (const event of readRoutingObservations()) {
65
+ const worker = wanted.get(event.selected);
66
+ if (!worker)
67
+ continue;
68
+ // Capability evidence belongs to the provider/model that produced it. This
69
+ // prevents a reused worker id from inheriting scores from a retired model.
70
+ if (event.provider !== worker.agent
71
+ || event.requestedModel !== worker.model
72
+ || event.requestedReasoningEffort !== worker.reasoningEffort)
73
+ continue;
74
+ if (typeof event.verificationSuccess !== 'boolean')
75
+ continue;
76
+ if (Date.parse(event.recordedAt) < oldestUseful)
77
+ continue;
78
+ grouped.set(event.selected, [...(grouped.get(event.selected) ?? []), event]);
79
+ }
80
+ const result = new Map();
81
+ for (const [id, events] of grouped) {
82
+ const successes = events.filter(event => event.verificationSuccess).length;
83
+ result.set(id, {
84
+ runs: events.length,
85
+ successes,
86
+ successRate: events.length ? successes / events.length : 0,
87
+ averageDurationMs: Math.round(events.reduce((sum, event) => sum + event.workerDurationMs, 0) / events.length),
88
+ averageTokens: Math.round(events.reduce((sum, event) => sum + event.inputTokens + event.outputTokens, 0) / events.length),
89
+ });
90
+ }
91
+ return result;
92
+ }
@@ -0,0 +1,195 @@
1
+ import { buildWatchdogInvocation, makeRunner, runCapturedAgent, runnerInvocation, } from '../loop/runner.js';
2
+ import { historyForWorkers, projectHash, recordRoutingObservation, storyHash } from './registry.js';
3
+ const costRank = { low: 0, medium: 1, high: 2 };
4
+ export function rankWorkers(workers, strategy, maxCandidates) {
5
+ const history = historyForWorkers(workers);
6
+ const ranked = [...workers].sort((a, b) => {
7
+ const ah = history.get(a.id);
8
+ const bh = history.get(b.id);
9
+ if (strategy === 'quality') {
10
+ const aq = ah?.successRate ?? 0.5;
11
+ const bq = bh?.successRate ?? 0.5;
12
+ return bq - aq || costRank[b.costTier] - costRank[a.costTier] || a.id.localeCompare(b.id);
13
+ }
14
+ if (strategy === 'speed') {
15
+ const ad = ah?.averageDurationMs ?? costRank[a.costTier] * 1_000_000;
16
+ const bd = bh?.averageDurationMs ?? costRank[b.costTier] * 1_000_000;
17
+ return ad - bd || costRank[a.costTier] - costRank[b.costTier] || a.id.localeCompare(b.id);
18
+ }
19
+ if (strategy === 'cost')
20
+ return costRank[a.costTier] - costRank[b.costTier] || a.id.localeCompare(b.id);
21
+ const as = (ah?.successRate ?? 0.75) * 10 - costRank[a.costTier];
22
+ const bs = (bh?.successRate ?? 0.75) * 10 - costRank[b.costTier];
23
+ return bs - as || a.id.localeCompare(b.id);
24
+ });
25
+ return ranked.slice(0, Math.max(1, maxCandidates));
26
+ }
27
+ export function buildRoutingPrompt(ctx, workers, strategy) {
28
+ const history = historyForWorkers(workers);
29
+ const candidates = workers.map(worker => {
30
+ const observed = history.get(worker.id);
31
+ const evidence = observed
32
+ ? `observed=${observed.successes}/${observed.runs} gate-verified avg=${observed.averageDurationMs}ms`
33
+ : 'observed=unproven';
34
+ return `- ${worker.id}: provider=${worker.agent}; cost=${worker.costTier}; capabilities=${worker.capabilities.join(',') || 'general'}; ${evidence}`;
35
+ });
36
+ return [
37
+ 'You are Yoke\'s routing controller. Choose who should execute one bounded coding story.',
38
+ 'Do not inspect files, use tools, implement code, or explain your reasoning at length.',
39
+ `Optimization strategy: ${strategy}. SELF is the strong parent and is appropriate when risk or ambiguity outweighs savings.`,
40
+ '',
41
+ `Story ${ctx.story.id}: ${ctx.story.title}`,
42
+ 'Acceptance criteria:',
43
+ ...ctx.story.acceptance.map(item => `- ${item}`),
44
+ '',
45
+ 'Allowed candidates:',
46
+ '- SELF: strong parent; highest confidence; highest expected cost',
47
+ ...candidates,
48
+ '',
49
+ 'Return exactly one line and nothing else:',
50
+ 'YOKE_ROUTE {"worker":"SELF-or-candidate-id","reason":"max 100 characters"}',
51
+ ].join('\n');
52
+ }
53
+ function allStrings(value, out) {
54
+ if (typeof value === 'string') {
55
+ out.push(value);
56
+ return;
57
+ }
58
+ if (Array.isArray(value)) {
59
+ for (const item of value)
60
+ allStrings(item, out);
61
+ return;
62
+ }
63
+ if (value && typeof value === 'object')
64
+ for (const item of Object.values(value))
65
+ allStrings(item, out);
66
+ }
67
+ export function parseRouteDecision(output, allowedWorkerIds) {
68
+ const strings = [output];
69
+ for (const line of output.split(/\r?\n/)) {
70
+ try {
71
+ allStrings(JSON.parse(line), strings);
72
+ }
73
+ catch { /* raw provider output is also searched */ }
74
+ }
75
+ const allowed = new Set(['SELF', ...allowedWorkerIds]);
76
+ for (const text of strings.reverse()) {
77
+ const match = text.match(/YOKE_ROUTE\s*(\{[^\r\n]*\})/);
78
+ if (!match)
79
+ continue;
80
+ try {
81
+ const value = JSON.parse(match[1]);
82
+ if (typeof value.worker !== 'string' || !allowed.has(value.worker))
83
+ continue;
84
+ return {
85
+ worker: value.worker,
86
+ reason: typeof value.reason === 'string' ? value.reason.slice(0, 100) : 'selected by orchestrator',
87
+ };
88
+ }
89
+ catch { /* try an earlier provider string */ }
90
+ }
91
+ return null;
92
+ }
93
+ function callUsage(role, provider, selection, tokens, durationMs, profile) {
94
+ return {
95
+ role,
96
+ provider,
97
+ ...(profile ? { profile } : {}),
98
+ ...(selection.model ? { requestedModel: selection.model } : {}),
99
+ ...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
100
+ ...(tokens?.model ? { actualModel: tokens.model } : {}),
101
+ inputTokens: tokens?.inputTokens ?? 0,
102
+ ...(tokens?.cachedInputTokens !== undefined ? { cachedInputTokens: tokens.cachedInputTokens } : {}),
103
+ ...(tokens?.cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens: tokens.cacheWriteInputTokens } : {}),
104
+ outputTokens: tokens?.outputTokens ?? 0,
105
+ ...(tokens?.reasoningOutputTokens !== undefined ? { reasoningOutputTokens: tokens.reasoningOutputTokens } : {}),
106
+ ...(tokens?.totalCostUsd !== undefined ? { totalCostUsd: tokens.totalCostUsd } : {}),
107
+ durationMs,
108
+ };
109
+ }
110
+ export function makeAdaptiveRunner(options) {
111
+ const now = options.now ?? Date.now;
112
+ const available = options.isAvailable ?? (() => true);
113
+ const eligibleWorkers = options.workers.filter(worker => available(worker.agent));
114
+ const makeWorker = options.makeWorker ?? ((agent, selection) => makeRunner(agent, options.idleTimeoutMs ?? 0, {
115
+ ...options.runnerOpts,
116
+ permissions: options.permissions ?? 'safe',
117
+ selection,
118
+ }));
119
+ return (ctx) => {
120
+ // Re-rank per story so a long-running loop can use gate outcomes learned by
121
+ // earlier stories without rebuilding the runner.
122
+ const candidates = rankWorkers(eligibleWorkers, options.strategy, options.maxCandidates);
123
+ if (candidates.length === 0) {
124
+ return makeWorker(options.parent, options.parentSelection ?? {})(ctx);
125
+ }
126
+ const prompt = buildRoutingPrompt(ctx, candidates, options.strategy);
127
+ const orchestratorSelection = { ...(options.parentSelection ?? {}), ...(options.orchestratorSelection ?? {}), nativeMultiAgent: false };
128
+ const orchestratorStarted = now();
129
+ const routeRun = options.captureRoute
130
+ ? options.captureRoute(options.parent, ctx, prompt, orchestratorSelection)
131
+ : runCapturedAgent(options.parent, buildWatchdogInvocation(runnerInvocation(options.parent, prompt, ctx.targetDir, true, 'read-only', orchestratorSelection), options.idleTimeoutMs ?? 0));
132
+ const orchestratorDurationMs = Math.max(0, now() - orchestratorStarted);
133
+ const decision = routeRun.success ? parseRouteDecision(routeRun.output, candidates.map(worker => worker.id)) : null;
134
+ const selected = decision?.worker ?? 'SELF';
135
+ const worker = selected === 'SELF' ? undefined : candidates.find(candidate => candidate.id === selected);
136
+ const provider = worker?.agent ?? options.parent;
137
+ const selection = worker
138
+ ? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, bare: options.parentSelection?.bare }
139
+ : { ...(options.parentSelection ?? {}), nativeMultiAgent: false };
140
+ const workerStarted = now();
141
+ const result = makeWorker(provider, selection)(ctx);
142
+ const workerDurationMs = Math.max(0, now() - workerStarted);
143
+ const calls = [
144
+ callUsage('orchestrator', options.parent, orchestratorSelection, routeRun.tokens, orchestratorDurationMs),
145
+ callUsage(worker ? 'worker' : 'parent', provider, selection, result.tokens, workerDurationMs, selected),
146
+ ];
147
+ const tokens = {
148
+ inputTokens: (routeRun.tokens?.inputTokens ?? 0) + (result.tokens?.inputTokens ?? 0),
149
+ ...((routeRun.tokens?.cachedInputTokens !== undefined || result.tokens?.cachedInputTokens !== undefined)
150
+ ? { cachedInputTokens: (routeRun.tokens?.cachedInputTokens ?? 0) + (result.tokens?.cachedInputTokens ?? 0) }
151
+ : {}),
152
+ ...((routeRun.tokens?.cacheWriteInputTokens !== undefined || result.tokens?.cacheWriteInputTokens !== undefined)
153
+ ? { cacheWriteInputTokens: (routeRun.tokens?.cacheWriteInputTokens ?? 0) + (result.tokens?.cacheWriteInputTokens ?? 0) }
154
+ : {}),
155
+ outputTokens: (routeRun.tokens?.outputTokens ?? 0) + (result.tokens?.outputTokens ?? 0),
156
+ ...((routeRun.tokens?.reasoningOutputTokens !== undefined || result.tokens?.reasoningOutputTokens !== undefined)
157
+ ? { reasoningOutputTokens: (routeRun.tokens?.reasoningOutputTokens ?? 0) + (result.tokens?.reasoningOutputTokens ?? 0) }
158
+ : {}),
159
+ ...((routeRun.tokens?.totalCostUsd !== undefined || result.tokens?.totalCostUsd !== undefined)
160
+ ? { totalCostUsd: (routeRun.tokens?.totalCostUsd ?? 0) + (result.tokens?.totalCostUsd ?? 0) }
161
+ : {}),
162
+ ...(result.tokens?.model ? { model: result.tokens.model } : routeRun.tokens?.model ? { model: routeRun.tokens.model } : {}),
163
+ calls,
164
+ };
165
+ let recorded = false;
166
+ const recordOutcome = (verificationSuccess) => {
167
+ if (recorded)
168
+ return;
169
+ recorded = true;
170
+ const project = projectHash(ctx.targetDir);
171
+ recordRoutingObservation({
172
+ projectHash: project,
173
+ storyHash: storyHash(project, ctx.story.id),
174
+ strategy: options.strategy,
175
+ selected,
176
+ provider,
177
+ ...(selection.model ? { requestedModel: selection.model } : {}),
178
+ ...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
179
+ ...(result.tokens?.model ? { actualModel: result.tokens.model } : {}),
180
+ orchestratorProvider: options.parent,
181
+ ...(orchestratorSelection.model ? { orchestratorModel: orchestratorSelection.model } : {}),
182
+ orchestratorDurationMs,
183
+ workerDurationMs,
184
+ processSuccess: result.success,
185
+ verificationSuccess,
186
+ inputTokens: tokens.inputTokens,
187
+ outputTokens: tokens.outputTokens,
188
+ });
189
+ };
190
+ const routeSummary = decision
191
+ ? `route=${selected} (${decision.reason})`
192
+ : `route=SELF (${routeRun.success ? 'invalid routing response' : 'orchestrator failed'})`;
193
+ return { ...result, summary: `${routeSummary}; ${result.summary}`, tokens, routing: { recordOutcome } };
194
+ };
195
+ }
@@ -5,6 +5,17 @@ import { loadConfig, saveConfig } from '../retrofit/config.js';
5
5
  import { detectProject } from '../retrofit/detect.js';
6
6
  import { runRetrofit } from '../retrofit/command.js';
7
7
  const ALL_AGENTS = ['claude', 'codex', 'gemini'];
8
+ export function defaultRoutingWorkers(agents) {
9
+ const workers = {
10
+ claude: { id: 'claude-fast', agent: 'claude', model: 'haiku', costTier: 'low', capabilities: ['exploration', 'mechanical-edits', 'tests'] },
11
+ // Inherit the account's current Codex model and lower only its supported effort.
12
+ // This avoids pinning a model id that will age out of the provider catalog.
13
+ codex: { id: 'codex-light', agent: 'codex', reasoningEffort: 'low', costTier: 'medium', capabilities: ['exploration', 'mechanical-edits', 'tests'] },
14
+ // Gemini CLI's default/Auto route tracks the models available to the active account.
15
+ gemini: { id: 'gemini-auto', agent: 'gemini', costTier: 'low', capabilities: ['large-context', 'exploration', 'implementation'] },
16
+ };
17
+ return agents.map(agent => workers[agent]).filter((worker) => worker !== undefined);
18
+ }
8
19
  function parseAgents(value, fallback) {
9
20
  if (value.trim().toLowerCase() === 'all')
10
21
  return [...ALL_AGENTS];
@@ -35,6 +46,7 @@ export async function runSetup(targetDir, opts = {}) {
35
46
  const defaultLoop = opts.loop ?? existing?.loop.enabled ?? true;
36
47
  const defaultRunner = opts.runner ?? existing?.runner?.agent ?? (host && defaultAgents.includes(host) ? host : defaultAgents[0] ?? host ?? 'claude');
37
48
  const defaultPolicy = opts.decisionPolicy ?? existing?.loop.decisionPolicy ?? (existing?.loop.onAmbiguity === 'abort' ? 'critical' : 'auto');
49
+ const defaultRouting = opts.routing ?? existing?.routing?.enabled ?? false;
38
50
  const interactive = opts.interactive ?? (process.stdin.isTTY === true && process.stdout.isTTY === true);
39
51
  let close;
40
52
  let ask = opts.ask;
@@ -49,6 +61,7 @@ export async function runSetup(targetDir, opts = {}) {
49
61
  let loop = defaultLoop;
50
62
  let runner = defaultRunner;
51
63
  let decisionPolicy = defaultPolicy;
64
+ let routing = defaultRouting;
52
65
  if (interactive && ask) {
53
66
  agents = parseAgents(await ask(`Agents [${defaultAgents.join(',')}] (claude,codex,gemini|all): `), defaultAgents);
54
67
  const graphAnswer = (await ask(`Code graph [${defaultGraph}] (graphify|serena): `)).trim().toLowerCase();
@@ -61,6 +74,7 @@ export async function runSetup(targetDir, opts = {}) {
61
74
  const policyAnswer = (await ask(`Decision mode [${decisionPolicy}] (auto|critical): `)).trim().toLowerCase();
62
75
  if (policyAnswer === 'auto' || policyAnswer === 'critical')
63
76
  decisionPolicy = policyAnswer;
77
+ routing = yes(await ask(`Enable adaptive multi-model routing? [${defaultRouting ? 'yes' : 'no'}]: `), defaultRouting);
64
78
  }
65
79
  if (!agents.includes(runner))
66
80
  agents = [...agents, runner];
@@ -72,8 +86,16 @@ export async function runSetup(targetDir, opts = {}) {
72
86
  return 1;
73
87
  config.loop = { ...config.loop, enabled: loop, decisionPolicy };
74
88
  config.runner = { ...config.runner, agent: runner };
89
+ const existingWorkers = config.routing?.workers ?? [];
90
+ config.routing = {
91
+ enabled: routing,
92
+ strategy: config.routing?.strategy ?? 'balanced',
93
+ maxCandidates: config.routing?.maxCandidates ?? 3,
94
+ ...(config.routing?.orchestrator ? { orchestrator: config.routing.orchestrator } : {}),
95
+ workers: existingWorkers.length > 0 ? existingWorkers : defaultRoutingWorkers(agents),
96
+ };
75
97
  saveConfig(targetDir, config);
76
- console.log(`Yoke setup complete: agents=${agents.join(',')} · runner=${runner} · loop=${loop ? 'on' : 'off'} · decisions=${decisionPolicy}`);
98
+ console.log(`Yoke setup complete: agents=${agents.join(',')} · runner=${runner} · loop=${loop ? 'on' : 'off'} · routing=${routing ? 'on' : 'off'} · decisions=${decisionPolicy}`);
77
99
  return 0;
78
100
  }
79
101
  finally {
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.1.0",
3
+ "version": "1.2.0",
4
4
  "description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
5
5
  "contextFileName": "GEMINI-EXTENSION.md"
6
6
  }