@hecer/yoke 1.20.0 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +46 -0
  4. package/README.md +6 -1
  5. package/bench/analyze-codex-comparison.mjs +90 -17
  6. package/bench/compare-codex.mjs +159 -36
  7. package/bench/result-schema.mjs +132 -0
  8. package/canon/manifest.yaml +1 -1
  9. package/dist/agents/pi-telemetry.js +2 -1
  10. package/dist/agents/process-streams.js +12 -64
  11. package/dist/agents/provider-selection.js +12 -0
  12. package/dist/agents/telemetry.js +52 -52
  13. package/dist/change/inbox.js +8 -3
  14. package/dist/check/command.js +69 -17
  15. package/dist/check/delivery.js +121 -0
  16. package/dist/cli.js +86 -8
  17. package/dist/code-intelligence/budgets.js +138 -0
  18. package/dist/code-intelligence/contracts.js +2 -0
  19. package/dist/code-intelligence/coordinator.js +156 -84
  20. package/dist/code-intelligence/evidence.js +87 -34
  21. package/dist/code-intelligence/mcp-client.js +10 -2
  22. package/dist/code-intelligence/mcp-server.js +10 -10
  23. package/dist/dashboard/analytics.js +5 -3
  24. package/dist/goals/command.js +183 -53
  25. package/dist/goals/usage.js +87 -0
  26. package/dist/loop/candidate-cleanup.js +47 -17
  27. package/dist/loop/candidates.js +17 -11
  28. package/dist/loop/cleanup.js +100 -0
  29. package/dist/loop/dispatcher.js +89 -26
  30. package/dist/loop/failure.js +104 -0
  31. package/dist/loop/gate-snapshot.js +19 -0
  32. package/dist/loop/git.js +1 -1
  33. package/dist/loop/loop.js +100 -66
  34. package/dist/loop/parallel-adapters.js +22 -4
  35. package/dist/loop/parallel-command.js +32 -5
  36. package/dist/loop/recovery.js +23 -5
  37. package/dist/loop/reporter.js +21 -4
  38. package/dist/loop/run-command.js +102 -48
  39. package/dist/loop/runner.js +5 -4
  40. package/dist/loop/worker.js +145 -91
  41. package/dist/observability/history.js +1 -1
  42. package/dist/observability/invocation.js +42 -0
  43. package/dist/observability/usage.js +17 -0
  44. package/dist/prd/command.js +20 -7
  45. package/dist/prd/decompose.js +5 -2
  46. package/dist/retrofit/command.js +22 -0
  47. package/dist/retrofit/config.js +26 -1
  48. package/dist/retrofit/gitignore.js +2 -0
  49. package/dist/retrofit/planners/claude.js +4 -4
  50. package/dist/retrofit/wsl.js +23 -3
  51. package/dist/routing/attempts.js +241 -0
  52. package/dist/routing/capability.js +13 -9
  53. package/dist/routing/optimization.js +73 -0
  54. package/dist/routing/registry.js +7 -1
  55. package/dist/routing/router.js +280 -127
  56. package/dist/setup/command.js +54 -6
  57. package/dist/smoke/command.js +302 -74
  58. package/docs/BENCHMARK-MANIFEST.md +131 -0
  59. package/docs/CODE-INTELLIGENCE.md +43 -1
  60. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  61. package/docs/DELIVERY-JOURNEYS.md +199 -0
  62. package/docs/ECONOMIC-ROUTING.md +180 -0
  63. package/docs/GOALS.md +61 -4
  64. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  65. package/docs/parallel-execution.md +37 -9
  66. package/gemini-extension.json +1 -1
  67. package/package.json +1 -1
@@ -1,7 +1,10 @@
1
- import { mkdirSync, rmSync, renameSync } from 'node:fs';
1
+ import { lstatSync, mkdirSync, rmSync, renameSync, writeFileSync } from 'node:fs';
2
2
  import { join, resolve } from 'node:path';
3
3
  import { createRequire } from 'node:module';
4
+ import { createHash } from 'node:crypto';
4
5
  import { loadConfig } from '../retrofit/config.js';
6
+ import { workspaceFingerprint } from '../workspace/fingerprint.js';
7
+ import { statePath } from '../workspace/state.js';
5
8
  const CONFIG_GUIDANCE = [
6
9
  'No smoke flows configured. Add a smoke section to .yoke/config.yaml, e.g.:',
7
10
  '',
@@ -24,7 +27,7 @@ export async function launchPlaywright(targetDir) {
24
27
  const chromium = pw.chromium ?? pw.default?.chromium;
25
28
  if (!chromium)
26
29
  return null;
27
- return await chromium.launch({ headless: true });
30
+ return await chromium.launch({ headless: true, timeout: 30_000 });
28
31
  }
29
32
  catch {
30
33
  return null;
@@ -32,7 +35,7 @@ export async function launchPlaywright(targetDir) {
32
35
  }
33
36
  // Flow names come from user config and become filenames — keep them safe.
34
37
  function safeName(name) {
35
- return name.replace(/[^\w.-]+/g, '-');
38
+ return name.replace(/[^\w.-]+/g, '-').slice(0, 120) || 'flow';
36
39
  }
37
40
  // The label names a directory that gets rmSync'd recursively — it must never
38
41
  // carry path semantics ('..', separators). Dots are stripped entirely so a
@@ -41,6 +44,204 @@ export function safeLabel(label) {
41
44
  const cleaned = label.replace(/[^\w-]+/g, '-').replace(/^-+|-+$/g, '');
42
45
  return cleaned || 'latest';
43
46
  }
47
+ class SmokeFailure extends Error {
48
+ code;
49
+ constructor(code) {
50
+ super(code);
51
+ this.code = code;
52
+ }
53
+ }
54
+ function failureCode(error, fallback) {
55
+ if (error instanceof SmokeFailure)
56
+ return error.code;
57
+ return error?.name === 'TimeoutError' ? 'timeout' : fallback;
58
+ }
59
+ async function bounded(operation, timeout) {
60
+ if (timeout <= 0)
61
+ throw new SmokeFailure('timeout');
62
+ const deadline = Date.now() + timeout;
63
+ let timer;
64
+ try {
65
+ const result = await Promise.race([
66
+ Promise.resolve().then(operation),
67
+ new Promise((_, reject) => { timer = setTimeout(() => reject(new SmokeFailure('timeout')), timeout); }),
68
+ ]);
69
+ if (Date.now() >= deadline)
70
+ throw new SmokeFailure('timeout');
71
+ return result;
72
+ }
73
+ finally {
74
+ if (timer)
75
+ clearTimeout(timer);
76
+ }
77
+ }
78
+ async function measured(stage, operation, fallback) {
79
+ const started = Date.now();
80
+ try {
81
+ await operation();
82
+ stage.status = 'passed';
83
+ }
84
+ catch (error) {
85
+ stage.status = 'failed';
86
+ stage.failure = failureCode(error, fallback);
87
+ throw new SmokeFailure(stage.failure);
88
+ }
89
+ finally {
90
+ stage.durationMs = Date.now() - started;
91
+ }
92
+ }
93
+ async function executeStep(page, step, baseUrl, timeout, fillValue) {
94
+ switch (step.action) {
95
+ case 'click':
96
+ if (!page.click)
97
+ throw new SmokeFailure('unsupported-action');
98
+ await page.click(step.selector, { timeout });
99
+ return;
100
+ case 'fill':
101
+ if (fillValue === undefined || fillValue.length > 8192)
102
+ throw new SmokeFailure('fill-value-unavailable');
103
+ if (!page.fill)
104
+ throw new SmokeFailure('unsupported-action');
105
+ await page.fill(step.selector, fillValue, { timeout });
106
+ return;
107
+ case 'press':
108
+ if (!page.press)
109
+ throw new SmokeFailure('unsupported-action');
110
+ await page.press(step.selector, step.key, { timeout });
111
+ return;
112
+ case 'expect-visible':
113
+ await page.waitForSelector(step.selector, { state: 'visible', timeout });
114
+ return;
115
+ case 'expect-text': {
116
+ if (!page.textContent)
117
+ throw new SmokeFailure('unsupported-action');
118
+ const deadline = Date.now() + timeout;
119
+ await page.waitForSelector(step.selector, { state: 'visible', timeout });
120
+ const normalize = (value) => value.replace(/\s+/gu, ' ').trim();
121
+ const expected = normalize(step.text);
122
+ // Both a fixed attempt bound and the enclosing deadline apply. No unbounded
123
+ // polling, even when a test adapter ignores Playwright's own timeout option.
124
+ for (let attempt = 0; attempt <= Math.ceil(timeout / 100); attempt++) {
125
+ const remaining = deadline - Date.now();
126
+ if (remaining <= 0)
127
+ break;
128
+ const actual = normalize(await page.textContent(step.selector, { timeout: remaining }) ?? '');
129
+ if (Date.now() < deadline && (step.exact ? actual === expected : actual.includes(expected)))
130
+ return;
131
+ await new Promise(resolveValue => setTimeout(resolveValue, Math.max(0, Math.min(100, deadline - Date.now()))));
132
+ }
133
+ throw new SmokeFailure('timeout');
134
+ }
135
+ case 'expect-url': {
136
+ if (!page.waitForURL)
137
+ throw new SmokeFailure('unsupported-action');
138
+ const expected = new URL(step.url, baseUrl).href;
139
+ await page.waitForURL(url => url.href === expected, { waitUntil: 'load', timeout });
140
+ return;
141
+ }
142
+ case 'reload': {
143
+ if (!page.reload)
144
+ throw new SmokeFailure('unsupported-action');
145
+ const response = await page.reload({ waitUntil: 'load', timeout });
146
+ if (response && !response.ok())
147
+ throw new SmokeFailure('http-error');
148
+ return;
149
+ }
150
+ default: {
151
+ const unexpected = step;
152
+ void unexpected;
153
+ throw new SmokeFailure('unsupported-action');
154
+ }
155
+ }
156
+ }
157
+ async function runFlow(browser, flow, baseUrl, proofDir, fileStem, fillValues, name) {
158
+ const started = Date.now(), deadline = started + (flow.timeoutMs ?? 60_000);
159
+ const report = {
160
+ name, status: 'failed', durationMs: 0, navigation: { status: 'skipped', durationMs: 0 },
161
+ ...(flow.landmark ? { landmark: { status: 'skipped', durationMs: 0 } } : {}),
162
+ steps: (flow.steps ?? []).map((step, index) => ({ index: index + 1, action: step.action, status: 'skipped', durationMs: 0 })),
163
+ };
164
+ let context, page, consoleErrors = 0;
165
+ let video = null;
166
+ const remaining = (limit) => Math.min(limit, deadline - Date.now());
167
+ const assertNoConsoleErrors = () => { if (consoleErrors > 0)
168
+ throw new SmokeFailure('console-errors'); };
169
+ try {
170
+ context = await bounded(() => browser.newContext({ recordVideo: { dir: join(proofDir, '.video-tmp') }, viewport: { width: 1280, height: 720 } }), remaining(30_000));
171
+ page = await bounded(() => context.newPage(), remaining(30_000));
172
+ const activePage = page;
173
+ // Browser diagnostics can echo input values. Count them; never persist their
174
+ // text, call logs, selectors or parameters in the machine-readable report.
175
+ page.on('console', msg => { if (msg?.type?.() === 'error')
176
+ consoleErrors++; });
177
+ page.on('pageerror', () => { consoleErrors++; });
178
+ await measured(report.navigation, async () => {
179
+ const timeout = remaining(30_000);
180
+ const response = await bounded(() => activePage.goto(baseUrl + flow.path, { waitUntil: 'load', timeout }), timeout);
181
+ if (response && !response.ok())
182
+ throw new SmokeFailure('http-error');
183
+ assertNoConsoleErrors();
184
+ }, 'navigation-failed');
185
+ if (flow.landmark && report.landmark) {
186
+ await measured(report.landmark, async () => {
187
+ const timeout = remaining(10_000);
188
+ await bounded(() => activePage.waitForSelector(flow.landmark, { state: 'visible', timeout }), timeout);
189
+ assertNoConsoleErrors();
190
+ }, 'landmark-missing');
191
+ }
192
+ for (const [index, step] of (flow.steps ?? []).entries()) {
193
+ await measured(report.steps[index], async () => {
194
+ const timeout = remaining(step.timeoutMs ?? 10_000);
195
+ await bounded(() => executeStep(activePage, step, baseUrl, timeout, fillValues.get(step)), timeout);
196
+ assertNoConsoleErrors();
197
+ }, 'step-failed');
198
+ }
199
+ }
200
+ catch (error) {
201
+ report.failure = failureCode(error, 'context-failed');
202
+ }
203
+ finally {
204
+ if (page) {
205
+ try {
206
+ await bounded(() => page.screenshot({ path: join(proofDir, `${fileStem}.png`), fullPage: true, timeout: 5000 }), 5000);
207
+ const saved = lstatSync(join(proofDir, `${fileStem}.png`));
208
+ if (!saved.isFile() || saved.size === 0)
209
+ throw new SmokeFailure('screenshot-failed');
210
+ report.screenshot = `${fileStem}.png`;
211
+ }
212
+ catch {
213
+ report.failure ??= 'screenshot-failed';
214
+ }
215
+ try {
216
+ video = page.video();
217
+ }
218
+ catch { /* Video is best-effort evidence. */ }
219
+ }
220
+ if (context) {
221
+ try {
222
+ await bounded(() => context.close(), 5000);
223
+ }
224
+ catch {
225
+ report.failure ??= 'context-cleanup-failed';
226
+ }
227
+ }
228
+ if (video) {
229
+ try {
230
+ const path = await bounded(() => video.path(), 5000);
231
+ if (report.failure) {
232
+ renameSync(path, join(proofDir, `${fileStem}.webm`));
233
+ report.failureVideo = `${fileStem}.webm`;
234
+ }
235
+ else
236
+ rmSync(path, { force: true });
237
+ }
238
+ catch { /* Preserve the original failure if video is unavailable. */ }
239
+ }
240
+ }
241
+ report.status = report.failure ? 'failed' : 'passed';
242
+ report.durationMs = Date.now() - started;
243
+ return report;
244
+ }
44
245
  export async function runFlowSmoke(targetDir, opts = {}) {
45
246
  const config = loadConfig(targetDir);
46
247
  const smoke = config?.smoke;
@@ -51,91 +252,118 @@ export async function runFlowSmoke(targetDir, opts = {}) {
51
252
  const baseUrl = opts.url ?? smoke.baseUrl;
52
253
  const label = safeLabel(opts.label ?? process.env.YOKE_STORY ?? 'latest');
53
254
  const proofRel = join('.yoke', 'proof', label);
54
- const proofDir = join(targetDir, proofRel);
255
+ let proofDir, sourceFingerprint;
256
+ try {
257
+ proofDir = statePath(targetDir, 'proof', label);
258
+ sourceFingerprint = workspaceFingerprint(targetDir);
259
+ }
260
+ catch {
261
+ console.error('Smoke evidence path or source fingerprint could not be verified; previous evidence was preserved.');
262
+ return 2;
263
+ }
264
+ const started = Date.now();
265
+ const fillValues = new Map();
266
+ for (const flow of smoke.flows)
267
+ for (const step of flow.steps ?? []) {
268
+ if (step.action === 'fill')
269
+ fillValues.set(step, step.value ?? (step.valueEnv ? process.env[step.valueEnv] : undefined));
270
+ }
271
+ const configDigest = createHash('sha256').update(JSON.stringify({ smoke, baseUrl, resolvedFillValues: [...fillValues.values()] })).digest('hex');
272
+ const redact = (text) => [...fillValues.values()].filter((value) => !!value).sort((a, b) => b.length - a.length).reduce((current, value) => current.split(value).join('[redacted]'), text);
55
273
  const launch = opts.launch ?? launchPlaywright;
56
- const browser = await launch(targetDir);
274
+ let browser;
275
+ try {
276
+ browser = await bounded(() => launch(targetDir), 30_000);
277
+ }
278
+ catch {
279
+ console.error('Smoke browser could not start; previous evidence was preserved.');
280
+ return 2;
281
+ }
57
282
  if (!browser) {
58
283
  console.error(`Playwright not found in ${targetDir}. Install it: npm i -D playwright && npx playwright install chromium`);
59
284
  return 2;
60
285
  }
61
286
  // Wipe only once the run is actually going to happen — an exit-2 run must
62
287
  // not destroy the previous run's evidence.
63
- rmSync(proofDir, { recursive: true, force: true }); // fresh evidence per run
64
- mkdirSync(proofDir, { recursive: true });
288
+ try {
289
+ statePath(targetDir, 'proof', label);
290
+ rmSync(proofDir, { recursive: true, force: true }); // fresh evidence per run
291
+ mkdirSync(proofDir, { recursive: true });
292
+ }
293
+ catch {
294
+ try {
295
+ await bounded(() => browser.close(), 5000);
296
+ }
297
+ catch { /* Report setup failed. */ }
298
+ console.error('Smoke evidence directory could not be prepared.');
299
+ return 2;
300
+ }
65
301
  const videoTmp = join(proofDir, '.video-tmp');
66
- let green = 0;
302
+ let baseOrigin = 'unavailable', browserVersion;
67
303
  try {
68
- for (const flow of smoke.flows) {
69
- const context = await browser.newContext({ recordVideo: { dir: videoTmp } });
70
- const page = await context.newPage();
71
- const errors = [];
72
- page.on('console', (msg) => {
73
- const m = msg;
74
- if (m.type?.() === 'error')
75
- errors.push(String(m.text?.() ?? msg));
76
- });
77
- page.on('pageerror', (err) => {
78
- errors.push(String(err?.message ?? err));
79
- });
80
- let reason;
81
- try {
82
- const resp = await page.goto(baseUrl + flow.path, { waitUntil: 'load', timeout: 30_000 });
83
- if (resp && !resp.ok())
84
- reason = `HTTP ${resp.status()}`;
85
- if (!reason && flow.landmark) {
86
- try {
87
- await page.waitForSelector(flow.landmark, { timeout: 10_000 });
88
- }
89
- catch {
90
- reason = `landmark "${flow.landmark}" not found`;
91
- }
92
- }
93
- if (!reason && errors.length > 0) {
94
- reason = `${errors.length} console error(s): ${errors[0].slice(0, 200)}`;
95
- }
96
- }
97
- catch (e) {
98
- reason = e.message.split('\n')[0];
99
- }
100
- // The screenshot IS the evidence — taken on success AND failure; a crashed
101
- // page must not mask the original failure.
102
- const shotName = `${safeName(flow.name)}.png`;
103
- let shotOk = false;
104
- try {
105
- await page.screenshot({ path: join(proofDir, shotName), fullPage: true });
106
- shotOk = true;
107
- }
108
- catch { /* keep the original reason */ }
109
- const video = page.video();
110
- await context.close();
111
- let videoKept = false;
112
- if (video) {
113
- try {
114
- const vpath = await video.path();
115
- if (reason) {
116
- renameSync(vpath, join(proofDir, `${safeName(flow.name)}.webm`));
117
- videoKept = true;
118
- }
119
- else {
120
- rmSync(vpath, { force: true });
121
- }
122
- }
123
- catch { /* video is best-effort evidence */ }
124
- }
125
- if (reason) {
126
- const saved = [shotOk ? 'screenshot' : null, videoKept ? 'video' : null].filter(Boolean).join(' + ');
127
- console.log(`✘ ${flow.name} — ${reason}${saved ? ` (${saved} saved under ${proofRel})` : ''}`);
128
- }
129
- else {
130
- green++;
131
- console.log(`✔ ${flow.name} (screenshot: ${join(proofRel, shotName)})`);
304
+ baseOrigin = new URL(baseUrl).origin;
305
+ }
306
+ catch { /* Navigation reports the invalid address. */ }
307
+ try {
308
+ const version = browser.version?.();
309
+ if (version)
310
+ browserVersion = redact(version.slice(0, 120));
311
+ }
312
+ catch { /* Optional runtime metadata. */ }
313
+ const report = {
314
+ version: 1, startedAt: new Date(started).toISOString(), generatedAt: '', durationMs: 0, status: 'failed',
315
+ sourceFingerprint, afterSourceFingerprint: null, sourceStable: false, configDigest,
316
+ environment: { platform: process.platform, architecture: process.arch, nodeVersion: process.version, driver: opts.launch ? 'injected' : 'playwright-chromium', ...(browserVersion ? { browserVersion } : {}), baseOrigin: redact(baseOrigin), viewport: { width: 1280, height: 720 } },
317
+ flows: [],
318
+ };
319
+ const filenames = new Set();
320
+ try {
321
+ for (const [index, flow] of smoke.flows.entries()) {
322
+ const baseName = safeName(redact(flow.name));
323
+ let fileStem = baseName, suffix = index + 1;
324
+ while (filenames.has(fileStem.toLowerCase()))
325
+ fileStem = `${baseName}-${suffix++}`;
326
+ filenames.add(fileStem.toLowerCase());
327
+ const result = await runFlow(browser, flow, baseUrl, proofDir, fileStem, fillValues, redact(flow.name));
328
+ report.flows.push(result);
329
+ if (result.status === 'failed') {
330
+ const saved = [result.screenshot ? 'screenshot' : null, result.failureVideo ? 'video' : null].filter(Boolean).join(' + ');
331
+ const failedStep = result.steps.find(step => step.status === 'failed');
332
+ console.log(`✘ ${result.name} — ${failedStep ? `step ${failedStep.index} (${failedStep.action}): ` : ''}${result.failure}${saved ? ` (${saved} saved under ${proofRel})` : ''}`);
132
333
  }
334
+ else
335
+ console.log(`✔ ${result.name} (screenshot: ${join(proofRel, result.screenshot)})`);
133
336
  }
134
337
  }
135
338
  finally {
136
- await browser.close();
339
+ try {
340
+ await bounded(() => browser.close(), 5000);
341
+ }
342
+ catch {
343
+ report.failure = 'browser-cleanup-failed';
344
+ }
137
345
  rmSync(videoTmp, { recursive: true, force: true });
138
346
  }
347
+ try {
348
+ report.afterSourceFingerprint = workspaceFingerprint(targetDir);
349
+ }
350
+ catch { /* Missing identity prevents a pass. */ }
351
+ report.sourceStable = report.sourceFingerprint === report.afterSourceFingerprint;
352
+ if (!report.sourceStable)
353
+ report.failure ??= 'source-changed-or-unavailable';
354
+ const green = report.flows.filter(flow => flow.status === 'passed').length;
355
+ report.status = !report.failure && green === smoke.flows.length ? 'passed' : 'failed';
356
+ report.generatedAt = new Date().toISOString();
357
+ report.durationMs = Date.now() - started;
358
+ try {
359
+ writeFileSync(statePath(targetDir, 'proof', label, 'report.json'), JSON.stringify(report, null, 2), { mode: 0o600 });
360
+ }
361
+ catch {
362
+ console.error('Smoke report could not be saved.');
363
+ return 2;
364
+ }
365
+ if (report.failure)
366
+ console.log(`✘ Flow-smoke evidence: ${report.failure}`);
139
367
  console.log(`Flow-smoke: ${green}/${smoke.flows.length} flows green — proof: ${proofRel}`);
140
- return green === smoke.flows.length ? 0 : 1;
368
+ return report.status === 'passed' ? 0 : 1;
141
369
  }
@@ -0,0 +1,131 @@
1
+ # Benchmark comparison manifests
2
+
3
+ Yoke 1.22.0 adds a versioned provenance and comparability contract to the direct
4
+ Codex comparison tools. This changes the tooling, not the historical results.
5
+ No authenticated development run or new savings measurement is implied.
6
+
7
+ ## What a new result records
8
+
9
+ The result document has schemaVersion 2 and a manifest with schemaVersion 1.
10
+ The manifest declares a unique comparison identity, its purpose (workflow or
11
+ controlled), the expected arms, the planned paired repeat count, and every
12
+ condition that is intentionally allowed to differ between arms. Each declared
13
+ difference needs a field name and a reason.
14
+
15
+ Each run contains its own context:
16
+
17
+ | Section | Recorded conditions |
18
+ | --- | --- |
19
+ | Source | Package version, Git commit, dirty state, SHA-256 build digest and whether source/build inputs stayed stable during the run |
20
+ | Fixture | Fixture identity and SHA-256 digests of the seed, original acceptance files and shared requirements |
21
+ | Execution | Provider, requested model, provider-reported model identities, effort, routing, native delegation, native goals, worker count and workflow |
22
+ | Prompt | Digest and scope of the submitted input: direct provider prompt or Yoke PRD/configuration bundle |
23
+ | Startup | Permissions, bare mode, ignore-rules, commit policy, isolation, timeout policy and user-state policy |
24
+ | Environment | Platform, architecture, Node version, provider CLI version, hashed host identity, background-load policy and whether skills/plugins were audited |
25
+
26
+ The build digest hashes the actual runtime files, canon, adapters/hooks, package
27
+ metadata, dependency lockfile and comparison tooling. It is recorded even for a
28
+ clean checkout, because an ignored dist directory can be stale. Dirty work is
29
+ therefore identified by both its base commit and runtime digest. Changes during a
30
+ run invalidate the stable-source condition. Missing Git or CLI provenance remains
31
+ unknown; it is never filled from a version written into an older report.
32
+
33
+ These digests identify recorded inputs. They are not signatures and do not prove
34
+ that a provider served a particular model. Reported model identity comes from
35
+ provider telemetry, never from the requested alias as a fallback.
36
+
37
+ ## Submitted inputs and effective prompts
38
+
39
+ The direct arm records a digest of the prompt sent to the provider. The Yoke arm
40
+ records a digest of the PRD and generated configuration submitted to the workflow.
41
+ Those inputs are expanded into story prompts by the separately hashed runtime.
42
+ The manifest labels these different scopes explicitly.
43
+
44
+ The tool does not claim to capture every dynamically expanded prompt, retry
45
+ message, provider system instruction, discovered skill or native context.
46
+ The seed and requirement digests still have to agree across all arms. A prompt
47
+ digest difference is permitted only when the manifest declares it.
48
+
49
+ ## Current workflow comparison
50
+
51
+ The comparison still measures one direct Codex session against Yoke's normal
52
+ story execution. Yoke can add worktrees, per-story provider starts, verification
53
+ and commits. These are deliberate workflow differences, not isolated measurements
54
+ of a scheduler or model-call overhead.
55
+
56
+ The existing direct-arm ignore-rules flag is retained and explicitly declared as
57
+ a startup-policy difference. Yoke retains its normal project rule handling.
58
+ Both arms request bare mode and disable native multi-agent delegation. The tool
59
+ does not silently modify global provider configuration or user rules.
60
+
61
+ Model and effort must match across arms unless a different experiment explicitly
62
+ declares those fields as variables. The same applies to actual model identities:
63
+ matching aliases alone do not establish a same-model comparison. An allowed
64
+ difference never makes a missing value valid. Fixture, requirement and acceptance
65
+ differences cannot be waived.
66
+
67
+ Conditions must stay stable within repeated runs of the same arm. A changing
68
+ adaptive model mix therefore needs a separately designed comparison contract;
69
+ this fixed-arm workflow tool does not silently pool such runs.
70
+
71
+ ## Analysis and output states
72
+
73
+ Run the analyzer on already recorded result files:
74
+
75
+ node bench/analyze-codex-comparison.mjs path/to/results.json
76
+
77
+ Files may contribute different arms of the same declared comparison. They must
78
+ agree on the manifest and supply every planned arm/repeat exactly once.
79
+ Different comparison identities are never pooled merely because their fixture
80
+ and arm labels match.
81
+
82
+ | State | Meaning |
83
+ | --- | --- |
84
+ | verified | All declared paired runs exist, recorded conditions match the contract, immutable acceptance passes, and input/cache/output telemetry is present |
85
+ | incompatible | Undeclared differences, changing within-arm conditions, conflicting manifests or duplicate measurements |
86
+ | incomplete | Some planned arms or repeats are missing |
87
+ | acceptance-failed | An arm did not pass immutable acceptance; its short failure time cannot count as a speedup |
88
+ | unverified | Missing/unknown provenance or telemetry, an unsupported schema, invalid context or an invalid variable declaration |
89
+ | legacy/unverified | The input predates manifests; its reported provenance and measurements remain readable but are not promoted to a verified comparison |
90
+
91
+ Only verified comparisons produce entries in the top-level groups array.
92
+ Other measurements remain in runs and explicitly labeled diagnostic summaries.
93
+ Missing or partial token measurements remain null. The analyzer does not invent
94
+ USD costs, CPU/memory measurements or savings percentages.
95
+
96
+ A successfully parsed diagnostic report can exit 0 while containing no verified
97
+ comparison. Consumers must inspect comparison status and groups. Malformed JSON,
98
+ impossible cache accounting, invalid timing and empty measurement inputs fail
99
+ the command.
100
+
101
+ ## Historical records
102
+
103
+ The analyzer reads both older results arrays and the runs arrays in dated
104
+ aggregate reports. It preserves their reported version and commit under
105
+ legacyReports, marks them legacy/unverified and leaves performance groups empty.
106
+ It no longer hardcodes Yoke 1.19.0 or the September audit commit.
107
+
108
+ The September 29 comparison and September 30 corrected smoke remain historical
109
+ evidence with their original sample sizes and limitations. The latter has no
110
+ contemporaneous baseline and must not be combined with the former to claim a
111
+ new relative speedup.
112
+
113
+ ## Measurement limits and authorized execution
114
+
115
+ Manifest verification establishes recorded comparability, not statistical
116
+ confidence, complete environment isolation or general product quality.
117
+ Current fixtures have visible tests and are small. Host load is uncontrolled,
118
+ skills/plugins are not exhaustively audited, and the dependency lockfile does
119
+ not certify every installed dependency or provider-internal setting.
120
+
121
+ Wall time starts after fixture preparation and ends when the runner exits.
122
+ Independent acceptance duration is separate; summaries also report elapsed time
123
+ through that acceptance. Source/environment hashing and fixture setup are
124
+ outside that timed interval.
125
+
126
+ The comparison launcher makes real authenticated provider calls. Run it only
127
+ under the user's authorized experiment scope. It accepts explicit arms, repeats,
128
+ model and effort; its default result root is a short OS-temporary directory
129
+ whose logs are retained. Tooling tests use synthetic records and do not launch
130
+ providers. An existing authorization for exactly two Curiuma development runs
131
+ is not permission for an additional fixture matrix or model-routing study.
@@ -28,9 +28,51 @@ Install and pin these tools in the project or developer environment before enabl
28
28
 
29
29
  - Every request is tied to a workspace and a content-addressed snapshot of committed and uncommitted, non-sensitive files.
30
30
  - Paths are workspace-relative; traversal, symlinks, `.env`/key material and policy-excluded paths are rejected.
31
- - Read calls are bounded by timeout and output budgets. Results carry backend, version, source path, content hash, resolution and freshness.
31
+ - Read calls share a request deadline and a complete-response output budget. Results distinguish backend identity, source location, semantic resolution and unverified index freshness. See the limits and evidence contracts below.
32
32
  - Edits run in a disposable Yoke sandbox. Preview writes a durable plan and diff but requires an approval record created by the Yoke control layer.
33
33
  - Apply rechecks the snapshot, plan expiry, plan hash, an exclusive transaction lock and idempotency. It produces a managed worktree and does not commit or merge; the existing Yoke loop and gates remain authoritative for integration.
34
34
  - Network access is not needed by the facade. Backend installation and any backend-specific network behavior remain an explicit environment concern.
35
35
 
36
36
  For a host that has no native MCP support (currently Pi), use the normal Yoke loop and bounded CLI/skill adapter. Yoke does not invent a second MCP transport for that host.
37
+
38
+ ## Evidence and trace contracts
39
+
40
+ `code_trace` treats Serena's incoming reference list as references to the requested symbol. Each validated referencing symbol has an edge **to the requested target**. Two adjacent results never imply an edge between those results. Incoming implementation lists similarly use the `implements` relation. A bare symbol name is resolved to a unique path-bound Serena identity before requesting these relations; ambiguous targets remain unresolved.
41
+
42
+ The pinned Graft trace interface returns text. Until an explicit edge format is validated, its output is retained as structural candidates, without invented call edges. Incoming references cannot establish outgoing calls, imports or inheritance: unsupported direction/relation combinations return partial results and explain the missing contract. `require_resolved` excludes unresolved structural candidates; it does not turn an unsupported query into a resolved graph. Semantic trace results currently cover one hop. Requests for greater depth are marked unverified instead of pretending traversal was completed.
43
+
44
+ `frontier_remaining` counts known entries omitted by result limits, and `traversal_complete` is false when exhaustive traversal is unverified. Zero known omissions does not prove that no other references exist. Node limits preserve graph endpoints: returned edges never point at removed nodes.
45
+
46
+ The workspace snapshot is content-addressed and an explicitly supplied snapshot is checked against the current workspace before processing. This check does **not** prove that Graft's index, Graphify's graph or the language server has processed those exact file contents. The supported wire contracts do not provide an independently validated index-to-snapshot binding. Consequently backend evidence uses `freshness: "unknown"` and `content_hash: null`; an untrusted backend field claiming `current` is not accepted as proof. The facade no longer stamps a current workspace file hash onto potentially older evidence.
47
+
48
+ Recognized, path-bound Serena symbol records can have `resolution: "resolved"` while freshness remains unknown. Unrecognized records remain unresolved and cannot create symbols or edges. Read responses use `status: "partial"` and partial coverage for available backends because index freshness and exhaustive coverage are not established. A successful process call alone does not imply complete coverage. Missing backends remain separately identified in `backends_missing`.
49
+
50
+ Each retained evidence link has a matching `provenance.evidence_id`. This additive field allows callers to resolve evidence IDs and lets output reduction remove unused provenance without breaking retained links. No response cache is used; no stale evidence is reused under a new snapshot or query.
51
+
52
+ ## Output and time limits
53
+
54
+ The facade forwards all four configured limits:
55
+
56
+ ```yaml
57
+ codeIntelligence:
58
+ mode: shadow
59
+ limits:
60
+ tokenBudget: 2400
61
+ timeoutMs: 10000
62
+ maxBytes: 2000000
63
+ maxBackends: 3
64
+ ```
65
+
66
+ The effective output cap is the smaller of `maxBytes` and the effective token budget. A request's `token_budget`, where supported, can lower the configured cap, but cannot raise it. The default configured token budget is 2400. The budget covers the entire serialized JSON response in MCP `structuredContent`: result data, provenance, warnings, identifiers, coverage, errors and the metrics themselves. The outer JSON-RPC framing and the short MCP text summary are outside that response.
67
+
68
+ **Token accounting is deliberately conservative:** one UTF-8 byte consumes one budget unit. `metrics.result_tokens` is therefore an estimate with `token_count_kind: "estimated"` and `token_count_method: "utf8_bytes_conservative"`. It is not provider-measured usage, a tokenizer-specific exact count or a billing estimate. `metrics.returned_bytes` is the actual UTF-8 size of the serialized structured response, including the metrics. This replaces the earlier four-bytes-per-token approximation and makes the same numeric budget stricter; increase the configured budget explicitly if a workflow requires larger context, rather than treating the old number as an exact token allowance.
69
+
70
+ When an answer does not fit, text is shortened at Unicode-safe boundaries and lower-ranked entries are removed. Arrays and operation receipt fields keep their schema, retained graph edges have valid endpoints, and unused provenance is removed. The answer reports `partial`, adds a truncation warning and never claims complete coverage. Narrow the query or raise the configured budget to obtain omitted content; `next_cursor: null` does not imply a complete search.
71
+
72
+ A budget can be smaller than the mandatory response envelope. For example, 128 budget units cannot encode all required identifiers and error fields. Such requests fail with `BUDGET_EXCEEDED` **before a backend is invoked**. The small, fixed-shape error envelope is the only output-cap exception; its size is reported accurately, and it contains no backend payload or unbounded error text. Apply additionally reserves space for its entire concrete receipt, including all changed paths and a possible deadline warning, before starting a transaction. An insufficient receipt budget rejects the apply without mutating. A completed apply keeps its transaction receipt even if synchronous work crossed the deadline.
73
+
74
+ The effective deadline is the smaller of configured `timeoutMs` (10000 ms by default) and the request's `timeout_ms`/tool default. All backend calls, semantic target resolution and preview operations consume the same remaining time. MCP initialization and its subsequent tool call also share that allowance. A backend which ignores its timeout is closed, late results are discarded, and no subsequent backend starts after expiry. Backend shutdown has a separate bounded cleanup grace; synchronous filesystem snapshot/transaction operations cannot be preempted mid-operation, so the deadline is not an exact end-to-end wall-clock promise. A final deadline check after normalization and output reduction marks delayed results partial and keeps any completed operation's receipt. A preview needing the former 120000 ms tool default must also have a configured timeout of at least that size.
75
+
76
+ ## Validation scope
77
+
78
+ The regression suite uses model-free adapters and local synthetic MCP processes. It verifies reference direction and endpoints, unknown payload handling, honest freshness and coverage, aggregate output limits, Unicode boundaries, evidence link preservation, tiny-budget failures, limit propagation, and shared deadlines including initialization. It does not assert a successful run against installed authenticated backends or an exact provider-token saving.
@@ -25,3 +25,18 @@ See the [Goals/resource audit](GOALS-RESOURCE-AUDIT-2026-09-29.md) for implement
25
25
  Reproduce with `bench/compare-codex.mjs`, then `bench/analyze-codex-comparison.mjs`. Use fresh, short output roots on Windows. The new summary tests verify median arithmetic, cache subtraction, unknown usage, incompatible policies and invalid measurements.
26
26
 
27
27
  Provenance: this report is disclosed as AI-assisted. Read-only text scans cannot establish human authorship or verify proprietary keyed watermarks; cryptographic verification and signer trust remain unknown without the corresponding verifier and trust policy.
28
+
29
+ ## Later tooling update — 1.22.0
30
+
31
+ The measurements above remain the historical 1.19.0 results. The current analyzer
32
+ reads their aggregate JSON as legacy/unverified because those runs did not record
33
+ the new complete comparison manifest. It preserves the reported values and does
34
+ not manufacture missing provenance or reinterpret them as 1.22.0 measurements.
35
+
36
+ New runs record source/build, fixture, acceptance and submitted-input digests,
37
+ requested and reported model identities, and explicit startup conditions.
38
+ The historical direct-arm ignore-rules difference is now an expressly declared
39
+ workflow variable. Undeclared cross-arm differences, missing provenance and
40
+ failed acceptance do not produce performance comparison groups.
41
+ See [benchmark manifests](BENCHMARK-MANIFEST.md) for the contract and limitations.
42
+ This is a tooling change; no new authenticated comparison was performed for it.