@hecer/yoke 1.21.1 → 1.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +48 -0
  4. package/README.md +8 -1
  5. package/TODOS.md +6 -0
  6. package/bench/analyze-codex-comparison.mjs +90 -17
  7. package/bench/compare-codex.mjs +159 -36
  8. package/bench/result-schema.mjs +132 -0
  9. package/canon/manifest.yaml +1 -1
  10. package/canon/skills/visual-verification/SKILL.md +25 -2
  11. package/canon/tools/codex-rtk-hook.mjs +6 -16
  12. package/dist/agents/pi-telemetry.js +2 -1
  13. package/dist/agents/process-streams.js +12 -64
  14. package/dist/agents/provider-selection.js +12 -0
  15. package/dist/agents/telemetry.js +52 -52
  16. package/dist/change/inbox.js +8 -3
  17. package/dist/check/command.js +69 -17
  18. package/dist/check/delivery.js +121 -0
  19. package/dist/cli.js +91 -3
  20. package/dist/code-intelligence/adapters/mcp.js +1 -0
  21. package/dist/code-intelligence/budgets.js +138 -0
  22. package/dist/code-intelligence/contracts.js +2 -0
  23. package/dist/code-intelligence/coordinator.js +159 -85
  24. package/dist/code-intelligence/evidence.js +87 -34
  25. package/dist/code-intelligence/index.js +1 -0
  26. package/dist/code-intelligence/mcp-client.js +25 -6
  27. package/dist/code-intelligence/mcp-server.js +14 -11
  28. package/dist/code-intelligence/preflight.js +71 -0
  29. package/dist/dashboard/analytics.js +5 -3
  30. package/dist/goals/command.js +183 -53
  31. package/dist/goals/usage.js +87 -0
  32. package/dist/loop/cache-isolation.js +36 -0
  33. package/dist/loop/candidate-cleanup.js +47 -17
  34. package/dist/loop/candidates.js +17 -11
  35. package/dist/loop/dispatcher.js +89 -26
  36. package/dist/loop/failure.js +104 -0
  37. package/dist/loop/gate-snapshot.js +19 -0
  38. package/dist/loop/git.js +1 -1
  39. package/dist/loop/loop.js +124 -70
  40. package/dist/loop/parallel-adapters.js +57 -6
  41. package/dist/loop/parallel-command.js +49 -7
  42. package/dist/loop/proof-retention.js +70 -0
  43. package/dist/loop/recovery.js +23 -5
  44. package/dist/loop/reporter.js +22 -5
  45. package/dist/loop/run-command.js +101 -47
  46. package/dist/loop/runner.js +6 -5
  47. package/dist/loop/worker.js +152 -91
  48. package/dist/observability/history.js +2 -1
  49. package/dist/observability/invocation.js +42 -0
  50. package/dist/observability/local-report.js +120 -0
  51. package/dist/observability/usage.js +19 -0
  52. package/dist/prd/command.js +20 -7
  53. package/dist/prd/decompose.js +5 -2
  54. package/dist/retrofit/config.js +29 -2
  55. package/dist/retrofit/gitignore.js +12 -0
  56. package/dist/retrofit/planners/codex.js +20 -20
  57. package/dist/routing/attempts.js +241 -0
  58. package/dist/routing/capability.js +13 -9
  59. package/dist/routing/optimization.js +73 -0
  60. package/dist/routing/registry.js +7 -1
  61. package/dist/routing/router.js +282 -127
  62. package/dist/setup/command.js +8 -2
  63. package/dist/smoke/command.js +387 -85
  64. package/dist/update/check.js +1 -1
  65. package/docs/BENCHMARK-MANIFEST.md +131 -0
  66. package/docs/CODE-INTELLIGENCE.md +43 -1
  67. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  68. package/docs/DELIVERY-JOURNEYS.md +206 -0
  69. package/docs/ECONOMIC-ROUTING.md +180 -0
  70. package/docs/GOALS.md +61 -4
  71. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  72. package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
  73. package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
  74. package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
  75. package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
  76. package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
  77. package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
  78. package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
  79. package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
  80. package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
  81. package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
  82. package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
  83. package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
  84. package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
  85. package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
  86. package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
  87. package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
  88. package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
  89. package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
  90. package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
  91. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
  92. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
  93. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
  94. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
  95. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
  96. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
  97. package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
  98. package/docs/parallel-execution.md +37 -9
  99. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
  100. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
  101. package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
  102. package/gemini-extension.json +1 -1
  103. package/package.json +1 -1
@@ -1,8 +1,11 @@
1
1
  import { existsSync, readFileSync, writeFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
+ import { execFileSync } from 'node:child_process';
3
4
  export const YOKE_IGNORE_LINES = [
5
+ '.yoke/supervision/',
4
6
  '.yoke/worktrees/',
5
7
  '.yoke/integration-recovery/',
8
+ '.yoke/failure-progress/',
6
9
  '.yoke/run-state.json',
7
10
  '.yoke/run-state.*.tmp',
8
11
  '.yoke/backup/',
@@ -31,6 +34,7 @@ export const YOKE_IGNORE_LINES = [
31
34
  '.yoke/checks/',
32
35
  '.yoke/events/',
33
36
  '.yoke/history/', '.yoke/routing/',
37
+ '.yoke/routing-attempts/',
34
38
  '.yoke/goal.json',
35
39
  '.yoke/goal.json.*.tmp',
36
40
  '.yoke/goal.pause',
@@ -40,6 +44,14 @@ const HEADER = '# Yoke runtime artifacts (managed by yoke retrofit)';
40
44
  // lines not already present (matched verbatim, line-wise). Preserves existing
41
45
  // content. Returns true if the file changed.
42
46
  export function ensureGitignore(targetDir) {
47
+ if (existsSync(join(targetDir, '.yoke', 'supervision'))) {
48
+ try {
49
+ const tracked = execFileSync('git', ['ls-files', '--', '.yoke/supervision/'], { cwd: targetDir, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }).trim();
50
+ if (tracked)
51
+ console.warn('Yoke supervision files are already tracked; ignore rules preserve their index entries. Review and explicitly untrack these runtime files before the next loop.');
52
+ }
53
+ catch { /* A retrofit may precede git initialization. */ }
54
+ }
43
55
  const file = join(targetDir, '.gitignore');
44
56
  const current = existsSync(file) ? readFileSync(file, 'utf8') : '';
45
57
  const present = new Set(current.split(/\r?\n/).map((l) => l.trim()));
@@ -1,8 +1,23 @@
1
- import { readFileSync } from 'node:fs';
1
+ import { existsSync, readFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
3
  import { loadManifest } from '../../canon/manifest.js';
4
4
  import { mcpServers, rtkInstruction } from '../tools.js';
5
5
  import { skillPackageActions } from '../skill-actions.js';
6
+ import { mergeJson } from '../merge-json.js';
7
+ function codexHooks(targetDir) {
8
+ const path = join(targetDir, '.codex/hooks.json');
9
+ const current = existsSync(path) ? JSON.parse(readFileSync(path, 'utf8')) : {};
10
+ // Remove only the exact Yoke adapter command; keep other hooks in its block.
11
+ if (Array.isArray(current.hooks?.PreToolUse)) {
12
+ current.hooks.PreToolUse = current.hooks.PreToolUse.flatMap((block) => {
13
+ if (!Array.isArray(block.hooks))
14
+ return [block];
15
+ const hooks = block.hooks.filter((hook) => hook.command !== 'node "$(git rev-parse --show-toplevel)/.codex/hooks/rtk.mjs"');
16
+ return hooks.length ? [{ ...block, hooks }] : [];
17
+ });
18
+ }
19
+ return JSON.stringify(mergeJson(current, { hooks: { PreToolUse: [{ matcher: 'Bash', hooks: [{ type: 'command', command: 'rtk hook codex' }] }] } }), null, 2) + '\n';
20
+ }
6
21
  function tomlMcp(codeGraph, codeIntelligence, targetDir) {
7
22
  const servers = mcpServers(codeGraph, codeIntelligence, targetDir);
8
23
  // Codex reads MCP servers from ~/.codex/config.toml. This project-level file is a
@@ -37,23 +52,8 @@ export function planCodex(canonDir, targetDir, codeGraph = 'graphify', codeIntel
37
52
  }, {
38
53
  kind: 'write',
39
54
  target: '.codex/hooks.json',
40
- merge: true,
41
- content: JSON.stringify({
42
- description: 'Yoke command compression for Codex',
43
- hooks: {
44
- PreToolUse: [{
45
- matcher: '^Bash$',
46
- hooks: [{
47
- type: 'command',
48
- command: 'node "$(git rev-parse --show-toplevel)/.codex/hooks/rtk.mjs"',
49
- commandWindows: 'powershell -NoProfile -ExecutionPolicy Bypass -Command "$root = git rev-parse --show-toplevel; node (Join-Path $root \'.codex/hooks/rtk.mjs\')"',
50
- timeout: 5,
51
- statusMessage: 'Compressing command output with RTK',
52
- }],
53
- }],
54
- },
55
- }, null, 2) + '\n',
56
- reason: 'rtk PreToolUse hook adapter',
55
+ content: codexHooks(targetDir),
56
+ reason: 'native RTK Codex hook; preserves foreign hooks and migrates the Yoke adapter',
57
57
  }, {
58
58
  kind: 'write',
59
59
  target: '.codex/hooks/rtk.mjs',
@@ -62,8 +62,8 @@ export function planCodex(canonDir, targetDir, codeGraph = 'graphify', codeIntel
62
62
  }, {
63
63
  kind: 'write',
64
64
  target: 'RTK.md',
65
- content: rtkInstruction() + '\n',
66
- reason: 'rtk instruction (Codex has no rewrite hook)',
65
+ content: rtkInstruction() + '\n\nCodex uses `rtk hook codex` when supported by the installed RTK. Run `yoke tools-preflight --json` to check native rewriting. Nested code-mode shell calls remain unverified; explicitly prefix verbose commands with RTK.\n',
66
+ reason: 'RTK guidance and operational verification',
67
67
  });
68
68
  for (const [name, description, sandbox, instructions] of roles) {
69
69
  actions.push({
@@ -0,0 +1,241 @@
1
+ import { createHash, randomUUID } from 'node:crypto';
2
+ import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
3
+ import { join } from 'node:path';
4
+ import { z } from 'zod';
5
+ import { statePath } from '../workspace/state.js';
6
+ const Hash = z.string().regex(/^[a-f0-9]{64}$/u);
7
+ const StoryHash = z.string().regex(/^[a-f0-9]{32}$/u);
8
+ const ReservationSchema = z.object({
9
+ version: z.literal(1), id: z.string(), contractKey: Hash, storyId: z.string().min(1),
10
+ ordinal: z.number().int().min(1).max(8), startedAt: z.string().datetime(),
11
+ limit: z.number().int().min(1).max(8), executionPolicyKey: z.string().optional(),
12
+ profile: z.string().min(1), provider: z.string().min(1),
13
+ requestedProvider: z.string().optional(), requestedModel: z.string().optional(),
14
+ requestedReasoningEffort: z.string().optional(), requestedVariant: z.string().optional(),
15
+ accountingScope: z.enum(['worker', 'execution-attempt']),
16
+ }).strict();
17
+ const OutcomeSchema = z.object({
18
+ version: z.literal(1), attemptId: z.string(), finishedAt: z.string().datetime(),
19
+ verificationSuccess: z.boolean(), failureKind: z.enum(['implementation', 'infrastructure']),
20
+ actualModel: z.string().optional(),
21
+ }).strict();
22
+ const UsageSchema = z.object({
23
+ version: z.literal(1), callId: z.string().min(1), role: z.string().min(1),
24
+ inputTokens: z.number().finite().nonnegative().optional(), outputTokens: z.number().finite().nonnegative().optional(),
25
+ totalCostUsd: z.number().finite().nonnegative().optional(), durationMs: z.number().finite().nonnegative().optional(),
26
+ usageAvailable: z.boolean(), costMeasurementComplete: z.boolean(),
27
+ }).strict();
28
+ const taskHash = (storyId) => createHash('sha256').update(storyId).digest('hex').slice(0, 32);
29
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
30
+ function directory(root, story, contract, create = false) {
31
+ StoryHash.parse(story);
32
+ Hash.parse(contract);
33
+ const parts = ['routing-attempts', story, contract];
34
+ for (let count = 0; count <= parts.length; count++) {
35
+ const path = statePath(root, ...parts.slice(0, count));
36
+ if (create && !existsSync(path))
37
+ mkdirSync(path);
38
+ if (existsSync(path) && !lstatSync(path).isDirectory())
39
+ throw Error('Routing attempt state is not a directory');
40
+ }
41
+ return statePath(root, ...parts);
42
+ }
43
+ function location(root, attemptId) {
44
+ const match = /^([a-f0-9]{32})\.([a-f0-9]{64})\.([1-8])$/u.exec(attemptId);
45
+ if (!match)
46
+ throw Error('Invalid routing attempt identity');
47
+ return { directory: directory(root, match[1], match[2]), ordinal: Number(match[3]) };
48
+ }
49
+ function readJson(file) {
50
+ const stat = lstatSync(file);
51
+ if (!stat.isFile() || stat.isSymbolicLink() || stat.size > 65_536)
52
+ throw Error('Invalid routing attempt state');
53
+ return JSON.parse(readFileSync(file, 'utf8'));
54
+ }
55
+ /** Publish complete JSON without replacing another process's reservation. The
56
+ * temporary file is flushed before the exclusive link makes it visible. */
57
+ function publishOnce(file, value) {
58
+ // dirname must also support Windows paths.
59
+ const safeTemp = file.replace(/[^\\/]+$/u, `.reservation-${randomUUID()}.tmp`);
60
+ let descriptor;
61
+ try {
62
+ descriptor = openSync(safeTemp, 'wx', 0o600);
63
+ writeFileSync(descriptor, JSON.stringify(value));
64
+ fsyncSync(descriptor);
65
+ closeSync(descriptor);
66
+ descriptor = undefined;
67
+ try {
68
+ linkSync(safeTemp, file);
69
+ return true;
70
+ }
71
+ catch (error) {
72
+ if (error.code === 'EEXIST')
73
+ return false;
74
+ throw error;
75
+ }
76
+ }
77
+ finally {
78
+ if (descriptor !== undefined)
79
+ closeSync(descriptor);
80
+ rmSync(safeTemp, { force: true });
81
+ }
82
+ }
83
+ export function readRoutingAttempts(root, storyId, contractKey) {
84
+ const dir = directory(root, taskHash(storyId), contractKey);
85
+ if (!existsSync(dir))
86
+ return [];
87
+ const entries = [];
88
+ for (const name of readdirSync(dir).filter(name => /^attempt-[1-8]\.json$/u.test(name)).sort()) {
89
+ const reservation = ReservationSchema.parse(readJson(join(dir, name)));
90
+ if (reservation.storyId !== storyId || reservation.contractKey !== contractKey || reservation.id !== `${taskHash(storyId)}.${contractKey}.${reservation.ordinal}`)
91
+ throw Error('Routing attempt identity does not match its task');
92
+ const outcomeFile = join(dir, `outcome-${reservation.ordinal}.json`);
93
+ const outcome = existsSync(outcomeFile) ? OutcomeSchema.parse(readJson(outcomeFile)) : undefined;
94
+ if (outcome && outcome.attemptId !== reservation.id)
95
+ throw Error('Routing outcome identity changed');
96
+ entries.push({ reservation, ...(outcome ? { outcome } : {}) });
97
+ }
98
+ return entries;
99
+ }
100
+ export function reserveRoutingAttempt(input) {
101
+ if (!Number.isSafeInteger(input.limit) || input.limit < 1 || input.limit > 8)
102
+ throw Error('Invalid routing attempt limit');
103
+ // Corrupt state blocks admission; optional analytics never supplies the budget.
104
+ readRoutingAttempts(input.root, input.storyId, input.contractKey);
105
+ const story = taskHash(input.storyId);
106
+ const dir = directory(input.root, story, input.contractKey, true);
107
+ for (let ordinal = 1; ordinal <= input.limit; ordinal++) {
108
+ const reservation = ReservationSchema.parse({
109
+ version: 1, id: `${story}.${input.contractKey}.${ordinal}`, contractKey: input.contractKey,
110
+ storyId: input.storyId, ordinal, startedAt: input.startedAt ?? new Date().toISOString(),
111
+ limit: input.limit, executionPolicyKey: input.executionPolicyKey,
112
+ profile: input.profile, provider: input.provider,
113
+ requestedProvider: input.selection?.provider, requestedModel: input.selection?.model,
114
+ requestedReasoningEffort: input.selection?.reasoningEffort, requestedVariant: input.selection?.variant,
115
+ accountingScope: input.accountingScope ?? 'worker',
116
+ });
117
+ if (publishOnce(join(dir, `attempt-${ordinal}.json`), reservation))
118
+ return reservation;
119
+ }
120
+ throw Error('Routing attempt budget exhausted; revise the task plan before retrying');
121
+ }
122
+ export function recordRoutingAttemptUsage(root, attemptId, usage) {
123
+ const point = location(root, attemptId);
124
+ if (!existsSync(join(point.directory, `attempt-${point.ordinal}.json`)))
125
+ throw Error('Routing usage has no admitted attempt');
126
+ const dir = join(point.directory, `usage-${point.ordinal}`);
127
+ if (existsSync(dir) && (lstatSync(dir).isSymbolicLink() || !lstatSync(dir).isDirectory()))
128
+ throw Error('Invalid routing usage directory');
129
+ if (!existsSync(dir))
130
+ mkdirSync(dir);
131
+ const explicitCalls = Boolean(usage.calls?.length);
132
+ const calls = explicitCalls ? usage.calls : [usage];
133
+ for (const call of calls) {
134
+ const callId = call.callId ?? usage.callId ?? randomUUID();
135
+ const complete = (!('usageAvailable' in call) || call.usageAvailable !== false) && ('measurementComplete' in call ? call.measurementComplete !== false : true)
136
+ && (explicitCalls || usage.measurementComplete !== false);
137
+ const entry = UsageSchema.parse({
138
+ version: 1, callId, role: call.role ?? usage.role ?? 'unknown',
139
+ ...(finite(call.inputTokens) ? { inputTokens: call.inputTokens } : {}),
140
+ ...(finite(call.outputTokens) ? { outputTokens: call.outputTokens } : {}),
141
+ ...(finite(call.totalCostUsd) ? { totalCostUsd: call.totalCostUsd } : {}),
142
+ ...(finite(call.durationMs) ? { durationMs: call.durationMs } : {}),
143
+ usageAvailable: complete && finite(call.inputTokens) && finite(call.outputTokens),
144
+ costMeasurementComplete: finite(call.totalCostUsd) && call.costMeasurementComplete !== false && (explicitCalls || usage.costMeasurementComplete !== false),
145
+ });
146
+ publishOnce(join(dir, `${createHash('sha256').update(callId).digest('hex')}.json`), entry);
147
+ }
148
+ }
149
+ export function markRoutingAttemptUsageIncomplete(root, attemptId) {
150
+ const point = location(root, attemptId);
151
+ publishOnce(join(point.directory, `incomplete-${point.ordinal}.json`), { version: 1 });
152
+ }
153
+ /** Join role usage only when the story has exactly one open execution. Parallel
154
+ * candidates require an explicit attempt id; ambiguity can never look complete. */
155
+ export function recordActiveRoutingUsage(root, storyId, usage) {
156
+ if (usage.routingAttemptId) {
157
+ recordRoutingAttemptUsage(root, usage.routingAttemptId, usage);
158
+ return true;
159
+ }
160
+ const story = taskHash(storyId);
161
+ const base = statePath(root, 'routing-attempts', story);
162
+ if (!existsSync(base))
163
+ return false;
164
+ if (!lstatSync(base).isDirectory())
165
+ throw Error('Invalid routing attempt directory');
166
+ const active = [];
167
+ const contracts = readdirSync(base).filter(name => /^[a-f0-9]{64}$/u.test(name));
168
+ if (contracts.length > 4096)
169
+ throw Error('Routing accounting history requires maintenance');
170
+ for (const contract of contracts) {
171
+ for (const entry of readRoutingAttempts(root, storyId, contract))
172
+ if (!entry.outcome)
173
+ active.push(entry.reservation);
174
+ }
175
+ if (active.length !== 1) {
176
+ for (const entry of active)
177
+ markRoutingAttemptUsageIncomplete(root, entry.id);
178
+ return false;
179
+ }
180
+ recordRoutingAttemptUsage(root, active[0].id, usage);
181
+ return true;
182
+ }
183
+ export function markActiveRoutingUsageIncomplete(root, storyId) {
184
+ const base = statePath(root, 'routing-attempts', taskHash(storyId));
185
+ if (!existsSync(base))
186
+ return;
187
+ const contracts = readdirSync(base).filter(name => /^[a-f0-9]{64}$/u.test(name));
188
+ for (const contract of contracts) {
189
+ for (const entry of readRoutingAttempts(root, storyId, contract))
190
+ if (!entry.outcome)
191
+ markRoutingAttemptUsageIncomplete(root, entry.reservation.id);
192
+ }
193
+ }
194
+ export function routingAttemptSummary(root, reservation, finishedAt = new Date().toISOString()) {
195
+ const point = location(root, reservation.id);
196
+ const dir = join(point.directory, `usage-${point.ordinal}`);
197
+ if (existsSync(dir) && (lstatSync(dir).isSymbolicLink() || !lstatSync(dir).isDirectory()))
198
+ throw Error('Invalid routing usage directory');
199
+ const names = existsSync(dir) ? readdirSync(dir).filter(name => /^[a-f0-9]{64}\.json$/u.test(name)) : [];
200
+ if (names.length > 10_000)
201
+ throw Error('Routing call ledger exceeds its limit');
202
+ const calls = names.map(name => UsageSchema.parse(readJson(join(dir, name))));
203
+ const ambiguous = existsSync(join(point.directory, `incomplete-${point.ordinal}.json`));
204
+ return {
205
+ calls: calls.length, usageComplete: !ambiguous && calls.length > 0 && calls.every(call => call.usageAvailable),
206
+ costComplete: !ambiguous && calls.length > 0 && calls.every(call => call.costMeasurementComplete),
207
+ ...(calls.some(call => call.totalCostUsd !== undefined) ? { totalCostUsd: calls.reduce((sum, call) => sum + (call.totalCostUsd ?? 0), 0) } : {}),
208
+ inputTokens: calls.reduce((sum, call) => sum + (call.inputTokens ?? 0), 0), outputTokens: calls.reduce((sum, call) => sum + (call.outputTokens ?? 0), 0),
209
+ durationMs: Math.max(0, Date.parse(finishedAt) - Date.parse(reservation.startedAt)), accountingScope: reservation.accountingScope,
210
+ };
211
+ }
212
+ export function finishRoutingAttempt(root, reservation, outcome) {
213
+ const point = location(root, reservation.id);
214
+ const file = join(point.directory, `outcome-${point.ordinal}.json`);
215
+ const result = OutcomeSchema.parse({ version: 1, attemptId: reservation.id, finishedAt: new Date().toISOString(), verificationSuccess: outcome.verificationSuccess, failureKind: outcome.failureKind ?? 'implementation', actualModel: outcome.actualModel });
216
+ publishOnce(file, result);
217
+ const stored = OutcomeSchema.parse(readJson(file));
218
+ return routingAttemptSummary(root, reservation, stored.finishedAt);
219
+ }
220
+ /** One economic observation covers the whole bounded execution sequence, charged
221
+ * to the starting profile. Escalation/repair spending is never made free by
222
+ * assigning it only to the final successful model. */
223
+ export function routingEpisodeSummary(root, reservation) {
224
+ const entries = readRoutingAttempts(root, reservation.storyId, reservation.contractKey);
225
+ const initial = entries[0];
226
+ const last = entries.at(-1);
227
+ const summaries = entries.map(entry => routingAttemptSummary(root, entry.reservation, entry.outcome?.finishedAt));
228
+ const complete = entries.every(entry => Boolean(entry.outcome))
229
+ && (last.outcome?.verificationSuccess === true || entries.length >= last.reservation.limit);
230
+ return {
231
+ id: `${taskHash(reservation.storyId)}.${reservation.contractKey}`,
232
+ initial: initial.reservation, actualInitialModel: initial.outcome?.actualModel,
233
+ complete, success: last.outcome?.verificationSuccess === true,
234
+ infrastructureFailure: entries.some(entry => entry.outcome?.failureKind === 'infrastructure'),
235
+ attempts: entries.length,
236
+ costComplete: complete && summaries.every(summary => summary.costComplete && summary.usageComplete && summary.accountingScope === 'execution-attempt')
237
+ && entries.every(entry => entry.reservation.executionPolicyKey === initial.reservation.executionPolicyKey),
238
+ ...(summaries.some(summary => summary.totalCostUsd !== undefined) ? { totalCostUsd: summaries.reduce((sum, summary) => sum + (summary.totalCostUsd ?? 0), 0) } : {}),
239
+ durationMs: summaries.reduce((sum, summary) => sum + summary.durationMs, 0),
240
+ };
241
+ }
@@ -3,6 +3,8 @@ import { join } from 'node:path';
3
3
  import { AssessmentSchema, assessmentKey, requiredTier, tiers } from './assessment.js';
4
4
  import { projectHash, readRoutingObservations } from './registry.js';
5
5
  import { currentContractKey } from './contracts.js';
6
+ import { readRoutingAttempts } from './attempts.js';
7
+ import { optimizeCapability } from './optimization.js';
6
8
  export function knownInfrastructureFailure(summary) {
7
9
  return /\bENOENT\b|\bECONNREFUSED\b|\bETIMEDOUT\b|CreateProcess(?:AsUserW|W)? failed|Failed to create unified exec process|command not found|is not recognized as|Missing script:|rate limit exceeded|authentication failed|AuthRequired|No access token was provided|invalid api key|credentials (?:missing|not found)|quota exceeded/i.test(summary);
8
10
  }
@@ -48,16 +50,17 @@ export function taskOutcomes(root, story) {
48
50
  }
49
51
  export function chooseCapability(input) {
50
52
  const role = input.role ?? 'implementation';
51
- const events = taskOutcomes(input.root, input.story);
52
- const lastSuccess = events.map(e => e.verificationSuccess).lastIndexOf(true);
53
- const failures = events.slice(lastSuccess + 1).filter(e => e.verificationSuccess === false);
53
+ const attempts = readRoutingAttempts(input.root, input.story.id, routingAssessmentKey(input.root, input.story));
54
+ const lastSuccess = attempts.map(entry => entry.outcome?.verificationSuccess).lastIndexOf(true);
55
+ const failures = attempts.slice(lastSuccess + 1).filter(entry => entry.outcome?.verificationSuccess === false && entry.outcome.failureKind !== 'infrastructure');
54
56
  const baseTier = requiredTier(input.assessment, role);
55
57
  const level = Math.min(3, tiers.indexOf(baseTier) + Math.max(0, failures.length - 1, (input.repairRound ?? 1) - 1));
56
- const exhausted = failures.length >= Math.min(input.maxAttempts ?? 5, 5 - tiers.indexOf(baseTier));
58
+ const attemptLimit = Math.min(input.maxAttempts ?? 5, 5 - tiers.indexOf(baseTier));
59
+ const exhausted = attempts.length >= attemptLimit;
57
60
  const candidates = input.workers.filter(w => w.tier && tiers.indexOf(w.tier) >= level && (!input.maxTier || tiers.indexOf(w.tier) <= tiers.indexOf(input.maxTier)) && (!w.roles || w.roles.includes(role)) && (!input.story.agent || w.agent === input.story.agent) && (input.available?.(w.agent) ?? true));
58
- const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && e.failureKind !== 'infrastructure' && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
61
+ const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
59
62
  const evidence = (w) => {
60
- const matching = history.filter(e => e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
63
+ const matching = history.filter(e => e.failureKind !== 'infrastructure' && e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
61
64
  const actual = matching.at(-1)?.actualModel;
62
65
  return actual ? matching.filter(e => e.actualModel === actual) : [];
63
66
  };
@@ -65,13 +68,14 @@ export function chooseCapability(input) {
65
68
  const reliable = candidates.filter(w => { const rows = evidence(w); return rows.length < 10 || rows.filter(e => e.verificationSuccess).length / rows.length >= 0.8; });
66
69
  const cost = { low: 0, medium: 1, high: 2 };
67
70
  reliable.sort((a, b) => tiers.indexOf(a.tier) - tiers.indexOf(b.tier) || cost[a.costTier] - cost[b.costTier] || a.id.localeCompare(b.id));
68
- const worker = reliable[0];
71
+ const economic = optimizeCapability({ workers: reliable, observations: history, assessment: input.assessment, settings: input.optimization, executionPolicyKey: input.executionPolicyKey, role });
72
+ const worker = economic.worker;
69
73
  const blocked = !worker && (input.fallback === 'block' || input.maxTier !== undefined);
70
74
  const provider = worker?.agent ?? input.story.agent ?? input.parent;
71
75
  const selection = worker ? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' && provider !== 'hermes' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
72
76
  : { ...(provider === input.parent ? input.parentSelection : {}), nativeMultiAgent: false };
73
- const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}`;
74
- return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
77
+ const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}${economic.reason ? `; ${economic.reason}` : ''}`;
78
+ return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, usedAttempts: attempts.length, attemptLimit, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
75
79
  }
76
80
  /** Explicit role models are resolved by callers before consulting this fallback. */
77
81
  export function roleSelection(root, config, story, provider, role, repairRound = 1) {
@@ -0,0 +1,73 @@
1
+ import { createHash } from 'node:crypto';
2
+ export function assessmentSignature(assessment) {
3
+ const { taskClass, difficulty, uncertainty, risk, scope, testability } = assessment;
4
+ return createHash('sha256').update(JSON.stringify({ taskClass, difficulty, uncertainty, risk, scope, testability })).digest('hex');
5
+ }
6
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
7
+ /** Compare completed, fully measured execution sequences, including failures and
8
+ * escalation, attributed to the initial profile. No benchmark probability, tier
9
+ * price estimate or incomplete successful-only sample is substituted for data. */
10
+ export function optimizeCapability(input) {
11
+ const baseline = input.workers[0];
12
+ if (!baseline || !input.settings)
13
+ return { worker: baseline, reason: '' };
14
+ if (input.role !== 'implementation')
15
+ return { worker: baseline, reason: 'economic selection unchanged: independent role-quality evidence is unavailable' };
16
+ if (!input.executionPolicyKey)
17
+ return { worker: baseline, reason: 'economic selection unchanged: comparable execution-policy evidence is unavailable' };
18
+ const required = Math.max(10, input.settings.minSamples ?? 20);
19
+ const signature = assessmentSignature(input.assessment);
20
+ const evidence = new Map();
21
+ for (const worker of input.workers) {
22
+ const bySequence = new Map();
23
+ for (const event of input.observations) {
24
+ const episode = event.economicEpisode;
25
+ if (!episode || !episode.initial || typeof episode.id !== 'string' || episode.assessmentSignature !== signature || episode.initial.executionPolicyKey !== input.executionPolicyKey)
26
+ continue;
27
+ const initial = episode.initial;
28
+ if (initial.profile !== worker.id || initial.provider !== worker.agent || initial.requestedProvider !== worker.provider
29
+ || initial.requestedModel !== worker.model || initial.requestedReasoningEffort !== worker.reasoningEffort || initial.requestedVariant !== worker.variant)
30
+ continue;
31
+ // Registry records are chronological; a later completion replaces the
32
+ // earlier open sequence instead of double-counting its repair attempts.
33
+ bySequence.delete(episode.id);
34
+ bySequence.set(episode.id, episode);
35
+ }
36
+ const rows = [...bySequence.values()];
37
+ const actual = rows.filter(row => row.actualInitialModel).at(-1)?.actualInitialModel;
38
+ const latest = rows.filter(row => !row.actualInitialModel || row.actualInitialModel === actual).slice(-required);
39
+ if (latest.length < required || latest.some(row => !row.complete || !row.costComplete || row.infrastructureFailure || !row.actualInitialModel
40
+ || row.initial.accountingScope !== 'execution-attempt' || !finite(row.totalCostUsd) || !finite(row.durationMs)))
41
+ continue;
42
+ const accepted = latest.filter(row => row.success).length;
43
+ if (!accepted)
44
+ continue;
45
+ evidence.set(worker.id, { samples: latest.length, accepted,
46
+ costPerAccepted: latest.reduce((sum, row) => sum + row.totalCostUsd, 0) / accepted,
47
+ durationPerAccepted: latest.reduce((sum, row) => sum + row.durationMs, 0) / accepted, });
48
+ }
49
+ const initial = evidence.get(baseline.id);
50
+ if (!initial)
51
+ return { worker: baseline, reason: `economic selection unchanged: need ${required} recent complete comparable execution sequences for the baseline; incomplete costs and unknown models are excluded from decisions` };
52
+ let selected = baseline, best = initial;
53
+ for (const worker of input.workers.slice(1)) {
54
+ const candidate = evidence.get(worker.id);
55
+ // Equal sample windows make the accepted counts comparable. Optimization may
56
+ // not trade away even the observed completion rate to obtain a lower bill.
57
+ if (!candidate || candidate.accepted < initial.accepted)
58
+ continue;
59
+ const improves = input.settings.objective === 'cost'
60
+ ? candidate.costPerAccepted < best.costPerAccepted
61
+ : input.settings.objective === 'speed'
62
+ ? candidate.durationPerAccepted < best.durationPerAccepted
63
+ : candidate.costPerAccepted <= best.costPerAccepted && candidate.durationPerAccepted <= best.durationPerAccepted
64
+ && (candidate.costPerAccepted < best.costPerAccepted || candidate.durationPerAccepted < best.durationPerAccepted);
65
+ if (improves) {
66
+ selected = worker;
67
+ best = candidate;
68
+ }
69
+ }
70
+ return { worker: selected,
71
+ reason: `economic ${input.settings.objective}: ${selected.id}; ${best.samples} fully measured comparable execution sequences, ${best.accepted}/${best.samples} accepted; observed $${best.costPerAccepted.toFixed(4)} and ${(best.durationPerAccepted / 1000).toFixed(1)}s per accepted sequence including failed attempts and escalation; observational evidence, not a calibrated forecast`,
72
+ };
73
+ }
@@ -45,7 +45,13 @@ export function readRoutingObservations(limit = 1000) {
45
45
  const dir = eventsDir();
46
46
  if (!existsSync(dir))
47
47
  return [];
48
- const files = readdirSync(dir).filter(file => file.endsWith('.json')).sort().slice(-limit);
48
+ let files;
49
+ try {
50
+ files = readdirSync(dir).filter(file => file.endsWith('.json')).sort().slice(-limit);
51
+ }
52
+ catch {
53
+ return [];
54
+ } // Optional evidence must never decide whether a task can run.
49
55
  const observations = [];
50
56
  for (const file of files) {
51
57
  try {