@hecer/yoke 1.21.1 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +27 -0
  4. package/README.md +6 -1
  5. package/bench/analyze-codex-comparison.mjs +90 -17
  6. package/bench/compare-codex.mjs +159 -36
  7. package/bench/result-schema.mjs +132 -0
  8. package/canon/manifest.yaml +1 -1
  9. package/dist/agents/pi-telemetry.js +2 -1
  10. package/dist/agents/process-streams.js +12 -64
  11. package/dist/agents/provider-selection.js +12 -0
  12. package/dist/agents/telemetry.js +52 -52
  13. package/dist/change/inbox.js +8 -3
  14. package/dist/check/command.js +69 -17
  15. package/dist/check/delivery.js +121 -0
  16. package/dist/cli.js +12 -3
  17. package/dist/code-intelligence/budgets.js +138 -0
  18. package/dist/code-intelligence/contracts.js +2 -0
  19. package/dist/code-intelligence/coordinator.js +156 -84
  20. package/dist/code-intelligence/evidence.js +87 -34
  21. package/dist/code-intelligence/mcp-client.js +10 -2
  22. package/dist/code-intelligence/mcp-server.js +10 -10
  23. package/dist/dashboard/analytics.js +5 -3
  24. package/dist/goals/command.js +183 -53
  25. package/dist/goals/usage.js +87 -0
  26. package/dist/loop/candidate-cleanup.js +47 -17
  27. package/dist/loop/candidates.js +17 -11
  28. package/dist/loop/dispatcher.js +89 -26
  29. package/dist/loop/failure.js +104 -0
  30. package/dist/loop/gate-snapshot.js +19 -0
  31. package/dist/loop/git.js +1 -1
  32. package/dist/loop/loop.js +100 -66
  33. package/dist/loop/parallel-adapters.js +22 -4
  34. package/dist/loop/parallel-command.js +32 -5
  35. package/dist/loop/recovery.js +23 -5
  36. package/dist/loop/reporter.js +21 -4
  37. package/dist/loop/run-command.js +95 -46
  38. package/dist/loop/runner.js +5 -4
  39. package/dist/loop/worker.js +145 -91
  40. package/dist/observability/history.js +1 -1
  41. package/dist/observability/invocation.js +42 -0
  42. package/dist/observability/usage.js +17 -0
  43. package/dist/prd/command.js +20 -7
  44. package/dist/prd/decompose.js +5 -2
  45. package/dist/retrofit/config.js +26 -1
  46. package/dist/retrofit/gitignore.js +2 -0
  47. package/dist/routing/attempts.js +241 -0
  48. package/dist/routing/capability.js +13 -9
  49. package/dist/routing/optimization.js +73 -0
  50. package/dist/routing/registry.js +7 -1
  51. package/dist/routing/router.js +280 -127
  52. package/dist/setup/command.js +8 -2
  53. package/dist/smoke/command.js +302 -74
  54. package/docs/BENCHMARK-MANIFEST.md +131 -0
  55. package/docs/CODE-INTELLIGENCE.md +43 -1
  56. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  57. package/docs/DELIVERY-JOURNEYS.md +199 -0
  58. package/docs/ECONOMIC-ROUTING.md +180 -0
  59. package/docs/GOALS.md +61 -4
  60. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  61. package/docs/parallel-execution.md +37 -9
  62. package/gemini-extension.json +1 -1
  63. package/package.json +1 -1
@@ -0,0 +1,42 @@
1
+ import { randomUUID } from 'node:crypto';
2
+ import { parseProviderTelemetry } from '../agents/telemetry.js';
3
+ import { appendEvent } from './events.js';
4
+ import { providerTelemetryUsage } from './usage.js';
5
+ /** A single invocation boundary, including failures and injected runner seams. */
6
+ export function measureInvocation(options) {
7
+ const started = Date.now(), callId = randomUUID(), runId = options.runId ?? randomUUID();
8
+ let result;
9
+ let failure;
10
+ const base = { runId, storyId: options.storyId, attemptId: callId, agent: options.agent, role: options.role };
11
+ appendEvent(options.root, { ...base, timestamp: new Date(started).toISOString(), type: 'status', data: { callId, parentCallId: options.parentCallId, state: 'started' } });
12
+ try {
13
+ result = options.execute(options.invocation);
14
+ return result;
15
+ }
16
+ catch (error) {
17
+ failure = error;
18
+ throw error;
19
+ }
20
+ finally {
21
+ let salvaged;
22
+ try {
23
+ const stdout = failure && typeof failure === 'object' && 'stdout' in failure ? failure.stdout : undefined;
24
+ salvaged = stdout == null ? undefined : providerTelemetryUsage(parseProviderTelemetry(options.agent, String(stdout).split(/\r?\n/u)));
25
+ }
26
+ catch { /* Failed measurement must not replace the original provider error. */ }
27
+ const usage = result?.tokens ?? salvaged;
28
+ const complete = usage !== undefined && usage.measurementComplete !== false;
29
+ const durationMs = Date.now() - started;
30
+ appendEvent(options.root, {
31
+ ...base, timestamp: new Date().toISOString(), type: 'tokens', durationMs, outcome: result?.success ? 'succeeded' : 'failed',
32
+ data: {
33
+ ...usage, callId, parentCallId: options.parentCallId,
34
+ provider: options.agent, role: options.role,
35
+ requestedProvider: options.selection?.provider, requestedModel: options.selection?.model,
36
+ requestedReasoningEffort: options.selection?.reasoningEffort, requestedVariant: options.selection?.variant,
37
+ actualModel: usage?.model, measurementComplete: complete, usageAvailable: complete,
38
+ costMeasurementComplete: usage?.costMeasurementComplete ?? (typeof usage?.totalCostUsd === 'number'),
39
+ },
40
+ });
41
+ }
42
+ }
@@ -0,0 +1,17 @@
1
+ /** Preserve known lower bounds; missing telemetry must never become a free call. */
2
+ export function providerTelemetryUsage(telemetry) {
3
+ if (!telemetry.tokens && !telemetry.partialUsage)
4
+ return undefined;
5
+ const known = { ...telemetry.partialUsage, ...telemetry.tokens };
6
+ if (!Object.values(known).some(value => typeof value === 'number'))
7
+ return undefined;
8
+ const model = known.model ?? (telemetry.reportedModels?.length === 1 ? telemetry.reportedModels[0] : undefined);
9
+ return {
10
+ ...known,
11
+ ...(model ? { model } : {}),
12
+ inputTokens: known.inputTokens ?? 0,
13
+ outputTokens: known.outputTokens ?? 0,
14
+ measurementComplete: telemetry.usageAvailable,
15
+ costMeasurementComplete: typeof telemetry.tokens?.totalCostUsd === 'number',
16
+ };
17
+ }
@@ -7,9 +7,11 @@ import { bindAssessments, preparedProblems } from './assess.js';
7
7
  import { resolvePlanner } from '../routing/planning.js';
8
8
  import { readPlanningFile } from '../routing/contracts.js';
9
9
  import { acquireLock, releaseLock } from '../loop/lock.js';
10
- import { agentInvocation, buildWatchdogInvocation, runAgent, isAgentAvailable, } from '../loop/runner.js';
10
+ import { agentInvocation, buildWatchdogInvocation, runCapturedAgent, isAgentAvailable, } from '../loop/runner.js';
11
11
  import { resolveIdleMs } from '../loop/run-command.js';
12
12
  import { detectHostAgent, resolveRunnerAgent } from '../agents/host.js';
13
+ import { measureInvocation } from '../observability/invocation.js';
14
+ import { statePath } from '../workspace/state.js';
13
15
  export const PRD_TEMPLATE = `# Yoke PRD — the loop picks the lowest-priority open story each iteration.
14
16
  # Story format (see canon/loop/prd.schema.md):
15
17
  # - id: STORY-1
@@ -92,14 +94,25 @@ export function runPrdDraft(targetDir, opts) {
92
94
  }
93
95
  try {
94
96
  const before = readPlanningFile(targetDir, '.yoke/prd.yaml');
95
- const rollback = () => { if (before === undefined)
96
- rmSync(path, { force: true });
97
- else
98
- writeFileSync(path, before); };
97
+ const rollback = () => {
98
+ const destination = join(statePath(targetDir), 'prd.yaml');
99
+ // Unlink a provider-created file link instead of writing through it.
100
+ rmSync(destination, { force: true });
101
+ if (before !== undefined)
102
+ writeFileSync(destination, before, { flag: 'wx' });
103
+ };
99
104
  const inv = agentInvocation(agent, buildPrdDraftPrompt(idea, planningBrief), targetDir, 'safe', planner.selection);
100
105
  console.log(`Drafting PRD with ${agent}...`);
101
- const run = opts.run ?? ((i) => runAgent(buildWatchdogInvocation(i, idleMs)));
102
- const result = run(inv);
106
+ const run = opts.run ?? ((i) => runCapturedAgent(agent, buildWatchdogInvocation(i, idleMs)));
107
+ let result;
108
+ try {
109
+ result = measureInvocation({ root: targetDir, agent, role: 'prd-draft', selection: planner.selection, invocation: inv, execute: run });
110
+ }
111
+ catch (error) {
112
+ rollback();
113
+ console.error(`PRD draft failed: ${error.message}`);
114
+ return 1;
115
+ }
103
116
  if (!result.success) {
104
117
  rollback();
105
118
  console.error(`PRD draft failed: ${result.summary}`);
@@ -11,7 +11,8 @@ import { writeScopesOverlap, validWriteScope } from '../loop/scheduler.js';
11
11
  import { withSharedWorkerSync } from '../loop/resource-pool.js';
12
12
  import { loadConfig } from '../retrofit/config.js';
13
13
  import { detectHostAgent, resolveRunnerAgent } from '../agents/host.js';
14
- import { agentInvocation, buildWatchdogInvocation, isAgentAvailable, runAgent } from '../loop/runner.js';
14
+ import { agentInvocation, buildWatchdogInvocation, isAgentAvailable, runCapturedAgent } from '../loop/runner.js';
15
+ import { measureInvocation } from '../observability/invocation.js';
15
16
  const ProposalSchema = z.array(z.object({
16
17
  id: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]{0,99}$/u),
17
18
  title: z.string().min(1).max(300),
@@ -181,7 +182,9 @@ export function runPrdDecompose(root, options) {
181
182
  if (prompt.length > 60_000)
182
183
  throw Error('Planning input exceeds 60000 characters; condense .yoke/plan.md first');
183
184
  const invocation = agentInvocation(planner.agent, prompt, root, 'safe', planner.selection);
184
- const result = withSharedWorkerSync({ targetDir: root, storyId: `prd-decompose:${parent.id}`, provider: planner.agent, role: 'implementation' }, () => (options.run ?? (item => runAgent(buildWatchdogInvocation(item, options.timeoutMinutes === undefined ? 20 * 60_000 : options.timeoutMinutes > 0 ? options.timeoutMinutes * 60_000 : 0))))(invocation));
185
+ const result = withSharedWorkerSync({ targetDir: root, storyId: `prd-decompose:${parent.id}`, provider: planner.agent, role: 'implementation' }, () => measureInvocation({ root, agent: planner.agent, role: 'prd-decompose', storyId: parent.id, selection: planner.selection, invocation,
186
+ execute: options.run ?? (item => runCapturedAgent(planner.agent, buildWatchdogInvocation(item, options.timeoutMinutes === undefined ? 20 * 60_000 : options.timeoutMinutes > 0 ? options.timeoutMinutes * 60_000 : 0))),
187
+ }));
185
188
  if (!result.success)
186
189
  throw Error(`planning request failed: ${result.summary}`);
187
190
  if (!existsSync(proposalPath))
@@ -16,7 +16,21 @@ const CodeIntelligenceSchema = z.object({
16
16
  policy: z.object({ allowNetwork: z.boolean().default(false), excludePatterns: z.array(z.string().min(1)).max(100).default([]), maxFileBytes: z.number().int().positive().max(20_000_000).optional() }).optional(),
17
17
  limits: z.object({ tokenBudget: z.number().int().min(128).max(16000).default(2400), timeoutMs: z.number().int().min(100).max(600000).default(10000), maxBytes: z.number().int().positive().max(10_000_000).default(2_000_000), maxBackends: z.number().int().min(1).max(3).default(3) }).optional(),
18
18
  });
19
- const SmokeFlowSchema = z.object({ name: z.string().min(1), path: z.string().min(1), landmark: z.string().optional() });
19
+ const SmokeStepTimeout = z.number().int().min(1).max(30000).optional();
20
+ const SmokeSelector = z.string().min(1).max(4096);
21
+ const SmokeStepSchema = z.discriminatedUnion('action', [
22
+ z.object({ action: z.literal('click'), selector: SmokeSelector, timeoutMs: SmokeStepTimeout }).strict(),
23
+ z.object({ action: z.literal('fill'), selector: SmokeSelector, value: z.string().max(8192).optional(), valueEnv: z.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/u).optional(), timeoutMs: SmokeStepTimeout }).strict(),
24
+ z.object({ action: z.literal('press'), selector: SmokeSelector, key: z.string().min(1).max(128), timeoutMs: SmokeStepTimeout }).strict(),
25
+ z.object({ action: z.literal('expect-visible'), selector: SmokeSelector, timeoutMs: SmokeStepTimeout }).strict(),
26
+ z.object({ action: z.literal('expect-text'), selector: SmokeSelector, text: z.string().max(8192), exact: z.boolean().optional(), timeoutMs: SmokeStepTimeout }).strict(),
27
+ z.object({ action: z.literal('expect-url'), url: z.string().min(1).max(4096), timeoutMs: SmokeStepTimeout }).strict(),
28
+ z.object({ action: z.literal('reload'), timeoutMs: SmokeStepTimeout }).strict(),
29
+ ]).superRefine((step, context) => {
30
+ if (step.action === 'fill' && (step.value === undefined) === (step.valueEnv === undefined))
31
+ context.addIssue({ code: 'custom', message: 'A fill step needs exactly one of value or valueEnv' });
32
+ });
33
+ const SmokeFlowSchema = z.object({ name: z.string().min(1), path: z.string().min(1), landmark: z.string().optional(), timeoutMs: z.number().int().min(1).max(120000).optional(), steps: z.array(SmokeStepSchema).min(1).max(50).optional() });
20
34
  const SmokeSchema = z.object({ baseUrl: z.string().min(1), flows: z.array(SmokeFlowSchema).min(1) });
21
35
  const OutputPolicySchema = z.object({
22
36
  previewBytes: z.number().int().positive().optional(),
@@ -56,6 +70,12 @@ const RoutingWorkerSchema = z.object({
56
70
  capabilities: z.array(z.string().min(1)).default([]),
57
71
  tier: z.enum(['light', 'standard', 'strong', 'frontier']).optional(),
58
72
  roles: z.array(z.enum(['implementation', 'reviewer', 'critic', 'repair'])).optional(),
73
+ profileMetadata: z.object({
74
+ version: z.literal(1),
75
+ catalog: z.string().min(1).max(120),
76
+ source: z.enum(['yoke-default', 'operator']),
77
+ basis: z.literal('configured-prior'),
78
+ }).strict().optional(),
59
79
  });
60
80
  const RoutingRuleSchema = z.object({
61
81
  area: z.string().min(1).optional(),
@@ -104,6 +124,11 @@ export const YokeConfigSchema = z.object({
104
124
  assessmentPolicy: z.enum(['on-demand', 'prepared']).optional(),
105
125
  fallback: z.enum(['parent', 'block']).optional(),
106
126
  maxTier: z.enum(['light', 'standard', 'strong', 'frontier']).optional(),
127
+ optimization: z.object({
128
+ version: z.literal(1),
129
+ objective: z.enum(['cost', 'speed', 'balanced']),
130
+ minSamples: z.number().int().min(10).max(10000).optional(),
131
+ }).strict().optional(),
107
132
  maxCandidates: z.number().int().min(1).max(5).default(3),
108
133
  orchestrator: z.object({
109
134
  provider: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/).optional(),
@@ -3,6 +3,7 @@ import { join } from 'node:path';
3
3
  export const YOKE_IGNORE_LINES = [
4
4
  '.yoke/worktrees/',
5
5
  '.yoke/integration-recovery/',
6
+ '.yoke/failure-progress/',
6
7
  '.yoke/run-state.json',
7
8
  '.yoke/run-state.*.tmp',
8
9
  '.yoke/backup/',
@@ -31,6 +32,7 @@ export const YOKE_IGNORE_LINES = [
31
32
  '.yoke/checks/',
32
33
  '.yoke/events/',
33
34
  '.yoke/history/', '.yoke/routing/',
35
+ '.yoke/routing-attempts/',
34
36
  '.yoke/goal.json',
35
37
  '.yoke/goal.json.*.tmp',
36
38
  '.yoke/goal.pause',
@@ -0,0 +1,241 @@
1
+ import { createHash, randomUUID } from 'node:crypto';
2
+ import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
3
+ import { join } from 'node:path';
4
+ import { z } from 'zod';
5
+ import { statePath } from '../workspace/state.js';
6
+ const Hash = z.string().regex(/^[a-f0-9]{64}$/u);
7
+ const StoryHash = z.string().regex(/^[a-f0-9]{32}$/u);
8
+ const ReservationSchema = z.object({
9
+ version: z.literal(1), id: z.string(), contractKey: Hash, storyId: z.string().min(1),
10
+ ordinal: z.number().int().min(1).max(8), startedAt: z.string().datetime(),
11
+ limit: z.number().int().min(1).max(8), executionPolicyKey: z.string().optional(),
12
+ profile: z.string().min(1), provider: z.string().min(1),
13
+ requestedProvider: z.string().optional(), requestedModel: z.string().optional(),
14
+ requestedReasoningEffort: z.string().optional(), requestedVariant: z.string().optional(),
15
+ accountingScope: z.enum(['worker', 'execution-attempt']),
16
+ }).strict();
17
+ const OutcomeSchema = z.object({
18
+ version: z.literal(1), attemptId: z.string(), finishedAt: z.string().datetime(),
19
+ verificationSuccess: z.boolean(), failureKind: z.enum(['implementation', 'infrastructure']),
20
+ actualModel: z.string().optional(),
21
+ }).strict();
22
+ const UsageSchema = z.object({
23
+ version: z.literal(1), callId: z.string().min(1), role: z.string().min(1),
24
+ inputTokens: z.number().finite().nonnegative().optional(), outputTokens: z.number().finite().nonnegative().optional(),
25
+ totalCostUsd: z.number().finite().nonnegative().optional(), durationMs: z.number().finite().nonnegative().optional(),
26
+ usageAvailable: z.boolean(), costMeasurementComplete: z.boolean(),
27
+ }).strict();
28
+ const taskHash = (storyId) => createHash('sha256').update(storyId).digest('hex').slice(0, 32);
29
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
30
+ function directory(root, story, contract, create = false) {
31
+ StoryHash.parse(story);
32
+ Hash.parse(contract);
33
+ const parts = ['routing-attempts', story, contract];
34
+ for (let count = 0; count <= parts.length; count++) {
35
+ const path = statePath(root, ...parts.slice(0, count));
36
+ if (create && !existsSync(path))
37
+ mkdirSync(path);
38
+ if (existsSync(path) && !lstatSync(path).isDirectory())
39
+ throw Error('Routing attempt state is not a directory');
40
+ }
41
+ return statePath(root, ...parts);
42
+ }
43
+ function location(root, attemptId) {
44
+ const match = /^([a-f0-9]{32})\.([a-f0-9]{64})\.([1-8])$/u.exec(attemptId);
45
+ if (!match)
46
+ throw Error('Invalid routing attempt identity');
47
+ return { directory: directory(root, match[1], match[2]), ordinal: Number(match[3]) };
48
+ }
49
+ function readJson(file) {
50
+ const stat = lstatSync(file);
51
+ if (!stat.isFile() || stat.isSymbolicLink() || stat.size > 65_536)
52
+ throw Error('Invalid routing attempt state');
53
+ return JSON.parse(readFileSync(file, 'utf8'));
54
+ }
55
+ /** Publish complete JSON without replacing another process's reservation. The
56
+ * temporary file is flushed before the exclusive link makes it visible. */
57
+ function publishOnce(file, value) {
58
+ // dirname must also support Windows paths.
59
+ const safeTemp = file.replace(/[^\\/]+$/u, `.reservation-${randomUUID()}.tmp`);
60
+ let descriptor;
61
+ try {
62
+ descriptor = openSync(safeTemp, 'wx', 0o600);
63
+ writeFileSync(descriptor, JSON.stringify(value));
64
+ fsyncSync(descriptor);
65
+ closeSync(descriptor);
66
+ descriptor = undefined;
67
+ try {
68
+ linkSync(safeTemp, file);
69
+ return true;
70
+ }
71
+ catch (error) {
72
+ if (error.code === 'EEXIST')
73
+ return false;
74
+ throw error;
75
+ }
76
+ }
77
+ finally {
78
+ if (descriptor !== undefined)
79
+ closeSync(descriptor);
80
+ rmSync(safeTemp, { force: true });
81
+ }
82
+ }
83
+ export function readRoutingAttempts(root, storyId, contractKey) {
84
+ const dir = directory(root, taskHash(storyId), contractKey);
85
+ if (!existsSync(dir))
86
+ return [];
87
+ const entries = [];
88
+ for (const name of readdirSync(dir).filter(name => /^attempt-[1-8]\.json$/u.test(name)).sort()) {
89
+ const reservation = ReservationSchema.parse(readJson(join(dir, name)));
90
+ if (reservation.storyId !== storyId || reservation.contractKey !== contractKey || reservation.id !== `${taskHash(storyId)}.${contractKey}.${reservation.ordinal}`)
91
+ throw Error('Routing attempt identity does not match its task');
92
+ const outcomeFile = join(dir, `outcome-${reservation.ordinal}.json`);
93
+ const outcome = existsSync(outcomeFile) ? OutcomeSchema.parse(readJson(outcomeFile)) : undefined;
94
+ if (outcome && outcome.attemptId !== reservation.id)
95
+ throw Error('Routing outcome identity changed');
96
+ entries.push({ reservation, ...(outcome ? { outcome } : {}) });
97
+ }
98
+ return entries;
99
+ }
100
+ export function reserveRoutingAttempt(input) {
101
+ if (!Number.isSafeInteger(input.limit) || input.limit < 1 || input.limit > 8)
102
+ throw Error('Invalid routing attempt limit');
103
+ // Corrupt state blocks admission; optional analytics never supplies the budget.
104
+ readRoutingAttempts(input.root, input.storyId, input.contractKey);
105
+ const story = taskHash(input.storyId);
106
+ const dir = directory(input.root, story, input.contractKey, true);
107
+ for (let ordinal = 1; ordinal <= input.limit; ordinal++) {
108
+ const reservation = ReservationSchema.parse({
109
+ version: 1, id: `${story}.${input.contractKey}.${ordinal}`, contractKey: input.contractKey,
110
+ storyId: input.storyId, ordinal, startedAt: input.startedAt ?? new Date().toISOString(),
111
+ limit: input.limit, executionPolicyKey: input.executionPolicyKey,
112
+ profile: input.profile, provider: input.provider,
113
+ requestedProvider: input.selection?.provider, requestedModel: input.selection?.model,
114
+ requestedReasoningEffort: input.selection?.reasoningEffort, requestedVariant: input.selection?.variant,
115
+ accountingScope: input.accountingScope ?? 'worker',
116
+ });
117
+ if (publishOnce(join(dir, `attempt-${ordinal}.json`), reservation))
118
+ return reservation;
119
+ }
120
+ throw Error('Routing attempt budget exhausted; revise the task plan before retrying');
121
+ }
122
+ export function recordRoutingAttemptUsage(root, attemptId, usage) {
123
+ const point = location(root, attemptId);
124
+ if (!existsSync(join(point.directory, `attempt-${point.ordinal}.json`)))
125
+ throw Error('Routing usage has no admitted attempt');
126
+ const dir = join(point.directory, `usage-${point.ordinal}`);
127
+ if (existsSync(dir) && (lstatSync(dir).isSymbolicLink() || !lstatSync(dir).isDirectory()))
128
+ throw Error('Invalid routing usage directory');
129
+ if (!existsSync(dir))
130
+ mkdirSync(dir);
131
+ const explicitCalls = Boolean(usage.calls?.length);
132
+ const calls = explicitCalls ? usage.calls : [usage];
133
+ for (const call of calls) {
134
+ const callId = call.callId ?? usage.callId ?? randomUUID();
135
+ const complete = (!('usageAvailable' in call) || call.usageAvailable !== false) && ('measurementComplete' in call ? call.measurementComplete !== false : true)
136
+ && (explicitCalls || usage.measurementComplete !== false);
137
+ const entry = UsageSchema.parse({
138
+ version: 1, callId, role: call.role ?? usage.role ?? 'unknown',
139
+ ...(finite(call.inputTokens) ? { inputTokens: call.inputTokens } : {}),
140
+ ...(finite(call.outputTokens) ? { outputTokens: call.outputTokens } : {}),
141
+ ...(finite(call.totalCostUsd) ? { totalCostUsd: call.totalCostUsd } : {}),
142
+ ...(finite(call.durationMs) ? { durationMs: call.durationMs } : {}),
143
+ usageAvailable: complete && finite(call.inputTokens) && finite(call.outputTokens),
144
+ costMeasurementComplete: finite(call.totalCostUsd) && call.costMeasurementComplete !== false && (explicitCalls || usage.costMeasurementComplete !== false),
145
+ });
146
+ publishOnce(join(dir, `${createHash('sha256').update(callId).digest('hex')}.json`), entry);
147
+ }
148
+ }
149
+ export function markRoutingAttemptUsageIncomplete(root, attemptId) {
150
+ const point = location(root, attemptId);
151
+ publishOnce(join(point.directory, `incomplete-${point.ordinal}.json`), { version: 1 });
152
+ }
153
+ /** Join role usage only when the story has exactly one open execution. Parallel
154
+ * candidates require an explicit attempt id; ambiguity can never look complete. */
155
+ export function recordActiveRoutingUsage(root, storyId, usage) {
156
+ if (usage.routingAttemptId) {
157
+ recordRoutingAttemptUsage(root, usage.routingAttemptId, usage);
158
+ return true;
159
+ }
160
+ const story = taskHash(storyId);
161
+ const base = statePath(root, 'routing-attempts', story);
162
+ if (!existsSync(base))
163
+ return false;
164
+ if (!lstatSync(base).isDirectory())
165
+ throw Error('Invalid routing attempt directory');
166
+ const active = [];
167
+ const contracts = readdirSync(base).filter(name => /^[a-f0-9]{64}$/u.test(name));
168
+ if (contracts.length > 4096)
169
+ throw Error('Routing accounting history requires maintenance');
170
+ for (const contract of contracts) {
171
+ for (const entry of readRoutingAttempts(root, storyId, contract))
172
+ if (!entry.outcome)
173
+ active.push(entry.reservation);
174
+ }
175
+ if (active.length !== 1) {
176
+ for (const entry of active)
177
+ markRoutingAttemptUsageIncomplete(root, entry.id);
178
+ return false;
179
+ }
180
+ recordRoutingAttemptUsage(root, active[0].id, usage);
181
+ return true;
182
+ }
183
+ export function markActiveRoutingUsageIncomplete(root, storyId) {
184
+ const base = statePath(root, 'routing-attempts', taskHash(storyId));
185
+ if (!existsSync(base))
186
+ return;
187
+ const contracts = readdirSync(base).filter(name => /^[a-f0-9]{64}$/u.test(name));
188
+ for (const contract of contracts) {
189
+ for (const entry of readRoutingAttempts(root, storyId, contract))
190
+ if (!entry.outcome)
191
+ markRoutingAttemptUsageIncomplete(root, entry.reservation.id);
192
+ }
193
+ }
194
+ export function routingAttemptSummary(root, reservation, finishedAt = new Date().toISOString()) {
195
+ const point = location(root, reservation.id);
196
+ const dir = join(point.directory, `usage-${point.ordinal}`);
197
+ if (existsSync(dir) && (lstatSync(dir).isSymbolicLink() || !lstatSync(dir).isDirectory()))
198
+ throw Error('Invalid routing usage directory');
199
+ const names = existsSync(dir) ? readdirSync(dir).filter(name => /^[a-f0-9]{64}\.json$/u.test(name)) : [];
200
+ if (names.length > 10_000)
201
+ throw Error('Routing call ledger exceeds its limit');
202
+ const calls = names.map(name => UsageSchema.parse(readJson(join(dir, name))));
203
+ const ambiguous = existsSync(join(point.directory, `incomplete-${point.ordinal}.json`));
204
+ return {
205
+ calls: calls.length, usageComplete: !ambiguous && calls.length > 0 && calls.every(call => call.usageAvailable),
206
+ costComplete: !ambiguous && calls.length > 0 && calls.every(call => call.costMeasurementComplete),
207
+ ...(calls.some(call => call.totalCostUsd !== undefined) ? { totalCostUsd: calls.reduce((sum, call) => sum + (call.totalCostUsd ?? 0), 0) } : {}),
208
+ inputTokens: calls.reduce((sum, call) => sum + (call.inputTokens ?? 0), 0), outputTokens: calls.reduce((sum, call) => sum + (call.outputTokens ?? 0), 0),
209
+ durationMs: Math.max(0, Date.parse(finishedAt) - Date.parse(reservation.startedAt)), accountingScope: reservation.accountingScope,
210
+ };
211
+ }
212
+ export function finishRoutingAttempt(root, reservation, outcome) {
213
+ const point = location(root, reservation.id);
214
+ const file = join(point.directory, `outcome-${point.ordinal}.json`);
215
+ const result = OutcomeSchema.parse({ version: 1, attemptId: reservation.id, finishedAt: new Date().toISOString(), verificationSuccess: outcome.verificationSuccess, failureKind: outcome.failureKind ?? 'implementation', actualModel: outcome.actualModel });
216
+ publishOnce(file, result);
217
+ const stored = OutcomeSchema.parse(readJson(file));
218
+ return routingAttemptSummary(root, reservation, stored.finishedAt);
219
+ }
220
+ /** One economic observation covers the whole bounded execution sequence, charged
221
+ * to the starting profile. Escalation/repair spending is never made free by
222
+ * assigning it only to the final successful model. */
223
+ export function routingEpisodeSummary(root, reservation) {
224
+ const entries = readRoutingAttempts(root, reservation.storyId, reservation.contractKey);
225
+ const initial = entries[0];
226
+ const last = entries.at(-1);
227
+ const summaries = entries.map(entry => routingAttemptSummary(root, entry.reservation, entry.outcome?.finishedAt));
228
+ const complete = entries.every(entry => Boolean(entry.outcome))
229
+ && (last.outcome?.verificationSuccess === true || entries.length >= last.reservation.limit);
230
+ return {
231
+ id: `${taskHash(reservation.storyId)}.${reservation.contractKey}`,
232
+ initial: initial.reservation, actualInitialModel: initial.outcome?.actualModel,
233
+ complete, success: last.outcome?.verificationSuccess === true,
234
+ infrastructureFailure: entries.some(entry => entry.outcome?.failureKind === 'infrastructure'),
235
+ attempts: entries.length,
236
+ costComplete: complete && summaries.every(summary => summary.costComplete && summary.usageComplete && summary.accountingScope === 'execution-attempt')
237
+ && entries.every(entry => entry.reservation.executionPolicyKey === initial.reservation.executionPolicyKey),
238
+ ...(summaries.some(summary => summary.totalCostUsd !== undefined) ? { totalCostUsd: summaries.reduce((sum, summary) => sum + (summary.totalCostUsd ?? 0), 0) } : {}),
239
+ durationMs: summaries.reduce((sum, summary) => sum + summary.durationMs, 0),
240
+ };
241
+ }
@@ -3,6 +3,8 @@ import { join } from 'node:path';
3
3
  import { AssessmentSchema, assessmentKey, requiredTier, tiers } from './assessment.js';
4
4
  import { projectHash, readRoutingObservations } from './registry.js';
5
5
  import { currentContractKey } from './contracts.js';
6
+ import { readRoutingAttempts } from './attempts.js';
7
+ import { optimizeCapability } from './optimization.js';
6
8
  export function knownInfrastructureFailure(summary) {
7
9
  return /\bENOENT\b|\bECONNREFUSED\b|\bETIMEDOUT\b|CreateProcess(?:AsUserW|W)? failed|Failed to create unified exec process|command not found|is not recognized as|Missing script:|rate limit exceeded|authentication failed|AuthRequired|No access token was provided|invalid api key|credentials (?:missing|not found)|quota exceeded/i.test(summary);
8
10
  }
@@ -48,16 +50,17 @@ export function taskOutcomes(root, story) {
48
50
  }
49
51
  export function chooseCapability(input) {
50
52
  const role = input.role ?? 'implementation';
51
- const events = taskOutcomes(input.root, input.story);
52
- const lastSuccess = events.map(e => e.verificationSuccess).lastIndexOf(true);
53
- const failures = events.slice(lastSuccess + 1).filter(e => e.verificationSuccess === false);
53
+ const attempts = readRoutingAttempts(input.root, input.story.id, routingAssessmentKey(input.root, input.story));
54
+ const lastSuccess = attempts.map(entry => entry.outcome?.verificationSuccess).lastIndexOf(true);
55
+ const failures = attempts.slice(lastSuccess + 1).filter(entry => entry.outcome?.verificationSuccess === false && entry.outcome.failureKind !== 'infrastructure');
54
56
  const baseTier = requiredTier(input.assessment, role);
55
57
  const level = Math.min(3, tiers.indexOf(baseTier) + Math.max(0, failures.length - 1, (input.repairRound ?? 1) - 1));
56
- const exhausted = failures.length >= Math.min(input.maxAttempts ?? 5, 5 - tiers.indexOf(baseTier));
58
+ const attemptLimit = Math.min(input.maxAttempts ?? 5, 5 - tiers.indexOf(baseTier));
59
+ const exhausted = attempts.length >= attemptLimit;
57
60
  const candidates = input.workers.filter(w => w.tier && tiers.indexOf(w.tier) >= level && (!input.maxTier || tiers.indexOf(w.tier) <= tiers.indexOf(input.maxTier)) && (!w.roles || w.roles.includes(role)) && (!input.story.agent || w.agent === input.story.agent) && (input.available?.(w.agent) ?? true));
58
- const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && e.failureKind !== 'infrastructure' && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
61
+ const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
59
62
  const evidence = (w) => {
60
- const matching = history.filter(e => e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
63
+ const matching = history.filter(e => e.failureKind !== 'infrastructure' && e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
61
64
  const actual = matching.at(-1)?.actualModel;
62
65
  return actual ? matching.filter(e => e.actualModel === actual) : [];
63
66
  };
@@ -65,13 +68,14 @@ export function chooseCapability(input) {
65
68
  const reliable = candidates.filter(w => { const rows = evidence(w); return rows.length < 10 || rows.filter(e => e.verificationSuccess).length / rows.length >= 0.8; });
66
69
  const cost = { low: 0, medium: 1, high: 2 };
67
70
  reliable.sort((a, b) => tiers.indexOf(a.tier) - tiers.indexOf(b.tier) || cost[a.costTier] - cost[b.costTier] || a.id.localeCompare(b.id));
68
- const worker = reliable[0];
71
+ const economic = optimizeCapability({ workers: reliable, observations: history, assessment: input.assessment, settings: input.optimization, executionPolicyKey: input.executionPolicyKey, role });
72
+ const worker = economic.worker;
69
73
  const blocked = !worker && (input.fallback === 'block' || input.maxTier !== undefined);
70
74
  const provider = worker?.agent ?? input.story.agent ?? input.parent;
71
75
  const selection = worker ? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' && provider !== 'hermes' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
72
76
  : { ...(provider === input.parent ? input.parentSelection : {}), nativeMultiAgent: false };
73
- const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}`;
74
- return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
77
+ const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}${economic.reason ? `; ${economic.reason}` : ''}`;
78
+ return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, usedAttempts: attempts.length, attemptLimit, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
75
79
  }
76
80
  /** Explicit role models are resolved by callers before consulting this fallback. */
77
81
  export function roleSelection(root, config, story, provider, role, repairRound = 1) {
@@ -0,0 +1,73 @@
1
+ import { createHash } from 'node:crypto';
2
+ export function assessmentSignature(assessment) {
3
+ const { taskClass, difficulty, uncertainty, risk, scope, testability } = assessment;
4
+ return createHash('sha256').update(JSON.stringify({ taskClass, difficulty, uncertainty, risk, scope, testability })).digest('hex');
5
+ }
6
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
7
+ /** Compare completed, fully measured execution sequences, including failures and
8
+ * escalation, attributed to the initial profile. No benchmark probability, tier
9
+ * price estimate or incomplete successful-only sample is substituted for data. */
10
+ export function optimizeCapability(input) {
11
+ const baseline = input.workers[0];
12
+ if (!baseline || !input.settings)
13
+ return { worker: baseline, reason: '' };
14
+ if (input.role !== 'implementation')
15
+ return { worker: baseline, reason: 'economic selection unchanged: independent role-quality evidence is unavailable' };
16
+ if (!input.executionPolicyKey)
17
+ return { worker: baseline, reason: 'economic selection unchanged: comparable execution-policy evidence is unavailable' };
18
+ const required = Math.max(10, input.settings.minSamples ?? 20);
19
+ const signature = assessmentSignature(input.assessment);
20
+ const evidence = new Map();
21
+ for (const worker of input.workers) {
22
+ const bySequence = new Map();
23
+ for (const event of input.observations) {
24
+ const episode = event.economicEpisode;
25
+ if (!episode || !episode.initial || typeof episode.id !== 'string' || episode.assessmentSignature !== signature || episode.initial.executionPolicyKey !== input.executionPolicyKey)
26
+ continue;
27
+ const initial = episode.initial;
28
+ if (initial.profile !== worker.id || initial.provider !== worker.agent || initial.requestedProvider !== worker.provider
29
+ || initial.requestedModel !== worker.model || initial.requestedReasoningEffort !== worker.reasoningEffort || initial.requestedVariant !== worker.variant)
30
+ continue;
31
+ // Registry records are chronological; a later completion replaces the
32
+ // earlier open sequence instead of double-counting its repair attempts.
33
+ bySequence.delete(episode.id);
34
+ bySequence.set(episode.id, episode);
35
+ }
36
+ const rows = [...bySequence.values()];
37
+ const actual = rows.filter(row => row.actualInitialModel).at(-1)?.actualInitialModel;
38
+ const latest = rows.filter(row => !row.actualInitialModel || row.actualInitialModel === actual).slice(-required);
39
+ if (latest.length < required || latest.some(row => !row.complete || !row.costComplete || row.infrastructureFailure || !row.actualInitialModel
40
+ || row.initial.accountingScope !== 'execution-attempt' || !finite(row.totalCostUsd) || !finite(row.durationMs)))
41
+ continue;
42
+ const accepted = latest.filter(row => row.success).length;
43
+ if (!accepted)
44
+ continue;
45
+ evidence.set(worker.id, { samples: latest.length, accepted,
46
+ costPerAccepted: latest.reduce((sum, row) => sum + row.totalCostUsd, 0) / accepted,
47
+ durationPerAccepted: latest.reduce((sum, row) => sum + row.durationMs, 0) / accepted, });
48
+ }
49
+ const initial = evidence.get(baseline.id);
50
+ if (!initial)
51
+ return { worker: baseline, reason: `economic selection unchanged: need ${required} recent complete comparable execution sequences for the baseline; incomplete costs and unknown models are excluded from decisions` };
52
+ let selected = baseline, best = initial;
53
+ for (const worker of input.workers.slice(1)) {
54
+ const candidate = evidence.get(worker.id);
55
+ // Equal sample windows make the accepted counts comparable. Optimization may
56
+ // not trade away even the observed completion rate to obtain a lower bill.
57
+ if (!candidate || candidate.accepted < initial.accepted)
58
+ continue;
59
+ const improves = input.settings.objective === 'cost'
60
+ ? candidate.costPerAccepted < best.costPerAccepted
61
+ : input.settings.objective === 'speed'
62
+ ? candidate.durationPerAccepted < best.durationPerAccepted
63
+ : candidate.costPerAccepted <= best.costPerAccepted && candidate.durationPerAccepted <= best.durationPerAccepted
64
+ && (candidate.costPerAccepted < best.costPerAccepted || candidate.durationPerAccepted < best.durationPerAccepted);
65
+ if (improves) {
66
+ selected = worker;
67
+ best = candidate;
68
+ }
69
+ }
70
+ return { worker: selected,
71
+ reason: `economic ${input.settings.objective}: ${selected.id}; ${best.samples} fully measured comparable execution sequences, ${best.accepted}/${best.samples} accepted; observed $${best.costPerAccepted.toFixed(4)} and ${(best.durationPerAccepted / 1000).toFixed(1)}s per accepted sequence including failed attempts and escalation; observational evidence, not a calibrated forecast`,
72
+ };
73
+ }
@@ -45,7 +45,13 @@ export function readRoutingObservations(limit = 1000) {
45
45
  const dir = eventsDir();
46
46
  if (!existsSync(dir))
47
47
  return [];
48
- const files = readdirSync(dir).filter(file => file.endsWith('.json')).sort().slice(-limit);
48
+ let files;
49
+ try {
50
+ files = readdirSync(dir).filter(file => file.endsWith('.json')).sort().slice(-limit);
51
+ }
52
+ catch {
53
+ return [];
54
+ } // Optional evidence must never decide whether a task can run.
49
55
  const observations = [];
50
56
  for (const file of files) {
51
57
  try {