@hecer/yoke 1.21.1 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +27 -0
  4. package/README.md +6 -1
  5. package/bench/analyze-codex-comparison.mjs +90 -17
  6. package/bench/compare-codex.mjs +159 -36
  7. package/bench/result-schema.mjs +132 -0
  8. package/canon/manifest.yaml +1 -1
  9. package/dist/agents/pi-telemetry.js +2 -1
  10. package/dist/agents/process-streams.js +12 -64
  11. package/dist/agents/provider-selection.js +12 -0
  12. package/dist/agents/telemetry.js +52 -52
  13. package/dist/change/inbox.js +8 -3
  14. package/dist/check/command.js +69 -17
  15. package/dist/check/delivery.js +121 -0
  16. package/dist/cli.js +12 -3
  17. package/dist/code-intelligence/budgets.js +138 -0
  18. package/dist/code-intelligence/contracts.js +2 -0
  19. package/dist/code-intelligence/coordinator.js +156 -84
  20. package/dist/code-intelligence/evidence.js +87 -34
  21. package/dist/code-intelligence/mcp-client.js +10 -2
  22. package/dist/code-intelligence/mcp-server.js +10 -10
  23. package/dist/dashboard/analytics.js +5 -3
  24. package/dist/goals/command.js +183 -53
  25. package/dist/goals/usage.js +87 -0
  26. package/dist/loop/candidate-cleanup.js +47 -17
  27. package/dist/loop/candidates.js +17 -11
  28. package/dist/loop/dispatcher.js +89 -26
  29. package/dist/loop/failure.js +104 -0
  30. package/dist/loop/gate-snapshot.js +19 -0
  31. package/dist/loop/git.js +1 -1
  32. package/dist/loop/loop.js +100 -66
  33. package/dist/loop/parallel-adapters.js +22 -4
  34. package/dist/loop/parallel-command.js +32 -5
  35. package/dist/loop/recovery.js +23 -5
  36. package/dist/loop/reporter.js +21 -4
  37. package/dist/loop/run-command.js +95 -46
  38. package/dist/loop/runner.js +5 -4
  39. package/dist/loop/worker.js +145 -91
  40. package/dist/observability/history.js +1 -1
  41. package/dist/observability/invocation.js +42 -0
  42. package/dist/observability/usage.js +17 -0
  43. package/dist/prd/command.js +20 -7
  44. package/dist/prd/decompose.js +5 -2
  45. package/dist/retrofit/config.js +26 -1
  46. package/dist/retrofit/gitignore.js +2 -0
  47. package/dist/routing/attempts.js +241 -0
  48. package/dist/routing/capability.js +13 -9
  49. package/dist/routing/optimization.js +73 -0
  50. package/dist/routing/registry.js +7 -1
  51. package/dist/routing/router.js +280 -127
  52. package/dist/setup/command.js +8 -2
  53. package/dist/smoke/command.js +302 -74
  54. package/docs/BENCHMARK-MANIFEST.md +131 -0
  55. package/docs/CODE-INTELLIGENCE.md +43 -1
  56. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  57. package/docs/DELIVERY-JOURNEYS.md +199 -0
  58. package/docs/ECONOMIC-ROUTING.md +180 -0
  59. package/docs/GOALS.md +61 -4
  60. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  61. package/docs/parallel-execution.md +37 -9
  62. package/gemini-extension.json +1 -1
  63. package/package.json +1 -1
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.21.1",
5
+ "version": "1.22.0",
6
6
  "description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi and Hermes: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.21.1",
3
+ "version": "1.22.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows for eight supported harnesses",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -1,5 +1,32 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.22.0 — 2026-10-03
4
+
5
+ ### Added
6
+ - Add durable routing attempt reservations independent of optional analytics retention, per-call goal admission and persistent consumption accounting, including planning and interrupted calls.
7
+ - Add optional evidence-based economic ranking inside capability routing. Cost, speed and balanced objectives require comparable, complete execution evidence; configured capability floors and fallback limits remain authoritative.
8
+ - Add `yoke goal assess` to prepare a bound goal contract explicitly, and record versioned provenance for setup model profiles as configuration priors rather than measured price or capability claims.
9
+ - Add compact persistent failure signatures: repeated unchanged failures request diagnosis and then stop instead of exhausting repeated identical turns. Completion failures carry typed reasons into repair planning.
10
+ - Add optional multi-step browser journeys and structured proof reports. Project checks can bind declared artifacts and journeys to an acceptance digest, source fingerprint, runtime environment and individual requirement results.
11
+ - Add versioned direct-Codex benchmark manifests containing actual source/build, fixture, acceptance, model and startup-policy provenance. Only compatible verified pairs contribute to comparisons; legacy data remains diagnostic.
12
+
13
+ ### Fixed
14
+ - Retain incomplete parallel work across failure, pause and decision boundaries, including unsuccessful candidate races, and resume implementation separately from already prepared integration candidates.
15
+ - Honor ambiguity aborts consistently in serial and parallel execution. Reuse successful gates only on unchanged source, task and gate inputs; rerun them after relevant changes and integration.
16
+ - Prevent exhausted routing budgets from resetting after statistics eviction or failed registry writes. Preserve the cost of planning even when routing blocks before implementation.
17
+ - Preserve known partial provider usage without converting unknown calls to zero-cost work or discarding measured worker usage. Record PRD drafting, decomposition and change-planning/coverage calls on success and failure.
18
+ - Apply goal assessment policy and routing rules, reset provider-specific defaults when changing runners, and separate historic native Codex bindings from the currently executing provider.
19
+ - Correct Code Intelligence reference edges, report unverified index freshness and traversal coverage honestly, and enforce shared request deadlines and bounded UTF-8 responses including evidence metadata.
20
+ - Synchronize package, lockfile, Canon, provider manifests and README release metadata.
21
+
22
+ ### Migration and validation limits
23
+ - Existing configurations retain their routing ranking unless `routing.optimization` is added. New setup configurations select `balanced` with a minimum of 20 comparable samples per candidate; insufficient or incomplete evidence keeps the conservative order. Profile tiers and `costTier` are configuration priors, not live provider prices.
24
+ - Goal token ceilings are checked before each additional call. A provider reporting only at call completion can still exceed a ceiling within that call; the overrun is retained and prevents further budgeted dispatch. Unknown interrupted consumption requires an explicit budget decision.
25
+ - Retained work consumes disk space until successful integration or explicit reviewed cleanup. Candidate recovery resumes the first retained alternative; additional alternatives remain available for inspection and cleanup. Custom lifecycles without `retain` keep their existing cleanup policy. Interrupted attempts remain spent budget and are excluded from model-quality comparisons; recovery does not invent missing accounting history.
26
+ - Delivery declarations map project-authored commands to artifacts and journeys. Hashes compare pre-check and post-check artifact content; they do not detect changes reverted between snapshots or independently prove that a command exercised a particular device or deployment. Artifact snapshots are limited to 512 MiB per file, 1 GiB total and 10 seconds per snapshot. Build artifacts before `yoke check`; inspect and explicitly refresh protected acceptance after intentional manifest changes.
27
+ - Code Intelligence token budgets use a documented conservative UTF-8 estimate, not provider-measured token counts. A tiny budget can only return a bounded `BUDGET_EXCEEDED` error envelope; backend freshness stays unknown without independent index evidence.
28
+ - This release uses deterministic regression and packaging checks. No new paid model benchmark or general speed/cost improvement is claimed; see the [validation record](docs/RELEASE-VALIDATION-1.22.0.md).
29
+
3
30
  ## 1.21.1 — 2026-10-03
4
31
 
5
32
  ### Fixed
package/README.md CHANGED
@@ -11,7 +11,7 @@
11
11
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
12
12
  ![Node.js 20+](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
13
13
 
14
- <!-- yoke:version:start -->1.21.1<!-- yoke:version:end --> · <!-- yoke:tests:start -->1426<!-- yoke:tests:end --> test cases · <!-- yoke:skills:start -->34<!-- yoke:skills:end --> skills
14
+ <!-- yoke:version:start -->1.22.0<!-- yoke:version:end --> · <!-- yoke:tests:start -->1610<!-- yoke:tests:end --> test cases · <!-- yoke:skills:start -->34<!-- yoke:skills:end --> skills
15
15
 
16
16
  <!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi | Hermes<!-- yoke:agents:end -->
17
17
 
@@ -41,6 +41,8 @@ Each story carries observable acceptance criteria and targeted test commands. Pa
41
41
  | **Verified completion** | Checks acceptance criteria, the project verify command, and any configured completion, review, quality, or browser gates before accepting work. |
42
42
  | **Parallel execution** | Schedules independent stories in isolated worktrees, respects dependencies and write scopes, and coordinates a shared worker limit across projects. |
43
43
  | **Durable goals** | Binds objectives to executable acceptance, shares worker capacity, records budgets, and optionally resumes native Codex goal threads. |
44
+ | **Measured routing** | Keeps hard attempt budgets independent of statistics, accounts for planning and reviews, and uses comparable complete evidence for optional economic model selection. |
45
+ | **Delivery evidence** | Records executable user journeys, acceptance results and stable declared build artifacts with their source and acceptance fingerprints. |
44
46
  | **Long-running autonomy** | Recovers from provider failures and blocked work. Optional exploration discovers new, repository-evidenced tasks after the planned backlog drains. |
45
47
  | **A choice of agents** | Uses the native CLI for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi, or Hermes. Install and authenticate the CLI you choose; Yoke does not bundle model runtimes or credentials. |
46
48
  | **Project visibility** | A local dashboard shows project status, **Workspace analytics**, **History**, and controls such as **Queue a change**, safe-boundary pause/resume, and operator notes. It runs with the local Yoke process. Dark and light themes are available. |
@@ -119,9 +121,12 @@ The dashboard is a local control room. It does not discover every process or run
119
121
  | [Continuous exploration](docs/CONTINUOUS-EXPLORATION.md) | Autonomous discovery, runtime limits, pause/resume, and stop detection |
120
122
  | [Project workflows](docs/VERIFIED-PROJECTS.md) | Setup, verification, goals, and execution defaults |
121
123
  | [Verified goals](docs/GOALS.md) | Acceptance binding, native Codex goals, budgets, and resource admission |
124
+ | [Economic routing](docs/ECONOMIC-ROUTING.md) | Durable attempt budgets, per-call costs, capability floors, and conservative model selection |
125
+ | [Delivery journeys](docs/DELIVERY-JOURNEYS.md) | Browser steps, project acceptance, artifact hashes, and proof limits |
122
126
  | [Code Intelligence](docs/CODE-INTELLIGENCE.md) | Optional Graphify, Serena, and Graft evidence providers |
123
127
  | [Dashboard](docs/DASHBOARD-EVOLUTION.md) | Project views, history, controls, and reporting boundaries |
124
128
  | [Benchmarks](bench/RESULTS.md) | Direct Codex comparison, routing studies, sample limits, and integration findings |
129
+ | [Benchmark manifests](docs/BENCHMARK-MANIFEST.md) | Comparable inputs, model identity, build provenance, and legacy-data limitations |
125
130
  | [Changelog](CHANGELOG.md) | Release features, behavior changes, and migration notes |
126
131
 
127
132
  ## Safety and limits
@@ -1,22 +1,95 @@
1
1
  // node bench/analyze-codex-comparison.mjs <results.json> ... > summary.json
2
2
  import { readFileSync } from 'node:fs'
3
+ import { assessComparison, stableComparisonJson } from './result-schema.mjs'
4
+
3
5
  const sources = process.argv.slice(2).map(path => JSON.parse(readFileSync(path, 'utf8')))
4
6
  if (!sources.length) throw new Error('Supply comparison results.json files')
5
- const rows = sources.flatMap(s => s.results).map(({ dir, ...row }) => ({
6
- ...row,
7
- freshInputTokens: Number.isFinite(row.usage?.inputTokens) && Number.isFinite(row.usage?.cachedInputTokens)
8
- ? row.usage.inputTokens - row.usage.cachedInputTokens : null,
7
+ const runs = []
8
+ const cohorts = new Map()
9
+ const legacyReports = []
10
+ const median = values => {
11
+ const sorted = [...values].sort((a, b) => a - b)
12
+ const n = sorted.length
13
+ return n ? (sorted[Math.floor((n - 1) / 2)] + sorted[Math.floor(n / 2)]) / 2 : null
14
+ }
15
+ const measured = value => Number.isFinite(value) && value >= 0
16
+ const metric = (rows, read) => rows.every(row => measured(read(row))) ? median(rows.map(read)) : null
17
+ function summarize(rows) {
18
+ return {
19
+ fixture: rows[0].fixture, arm: rows[0].arm, runs: rows.length,
20
+ accepted: rows.filter(row => row.accepted === true).length,
21
+ medianWallMs: median(rows.map(row => row.wallMs)),
22
+ medianElapsedToAcceptanceMs: metric(rows, row => measured(row.acceptanceMs) ? row.wallMs + row.acceptanceMs : null),
23
+ medianFreshInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.freshInputTokens),
24
+ medianOutputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.outputTokens),
25
+ medianAllInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.inputTokens),
26
+ }
27
+ }
28
+ function armSummaries(rows) {
29
+ return [...new Set(rows.map(row => row.arm))].map(arm => summarize(rows.filter(row => row.arm === arm)))
30
+ }
31
+ for (const [index, source] of sources.entries()) {
32
+ const input = source.results ?? source.runs
33
+ if (!Array.isArray(input) || !input.length) throw new Error('No completed runs in input ' + (index + 1))
34
+ const rows = input.map(({ dir, ...row }) => {
35
+ if (!measured(row.wallMs) || row.wallMs <= 0) throw new Error('Invalid measurement: wall time')
36
+ for (const field of ['inputTokens', 'cachedInputTokens', 'outputTokens']) {
37
+ if (row.usage?.[field] != null && !measured(row.usage[field])) throw new Error('Invalid measurement: ' + field)
38
+ }
39
+ if (row.acceptanceMs != null && !measured(row.acceptanceMs)) throw new Error('Invalid measurement: acceptance time')
40
+ const freshInputTokens = measured(row.usage?.inputTokens) && measured(row.usage?.cachedInputTokens)
41
+ ? row.usage.inputTokens - row.usage.cachedInputTokens : null
42
+ if (freshInputTokens !== null && freshInputTokens < 0) throw new Error('Invalid measurement: cache exceeds input')
43
+ return { ...row, freshInputTokens }
44
+ })
45
+ if (!source.manifest) {
46
+ legacyReports.push({
47
+ status: 'legacy/unverified', reportedVersion: source.version ?? source.auditedVersion ?? null,
48
+ reportedCommit: source.commit ?? source.auditedCommit ?? null, runs: rows.length,
49
+ reason: 'No versioned manifest; provenance and cross-arm conditions cannot be verified',
50
+ diagnosticSummaries: armSummaries(rows),
51
+ })
52
+ runs.push(...rows.map(row => ({ ...row, evidenceStatus: 'legacy/unverified' })))
53
+ continue
54
+ }
55
+ // A malformed identity remains its own unverified cohort, never joins another file.
56
+ const id = typeof source.manifest.id === 'string' && source.manifest.id ? source.manifest.id : 'invalid-input-' + index
57
+ const cohort = cohorts.get(id) ?? { id, manifests: [], runs: [] }
58
+ cohort.manifests.push(source.manifest)
59
+ cohort.runs.push(...rows)
60
+ cohorts.set(id, cohort)
61
+ }
62
+ const comparisons = [...cohorts.values()].map(cohort => ({
63
+ id: cohort.id,
64
+ ...assessComparison(cohort.manifests, cohort.runs),
65
+ manifest: cohort.manifests[0],
66
+ provenance: [...new Map(cohort.runs.map(row => [stableComparisonJson(row.context?.source), row.context?.source ?? null])).values()],
67
+ diagnosticSummaries: armSummaries(cohort.runs),
9
68
  }))
10
- if (!rows.length) throw new Error('No completed runs')
11
- for (const row of rows) if (!Number.isFinite(row.wallMs) || row.wallMs <= 0 || (row.freshInputTokens !== null && row.freshInputTokens < 0)) throw new Error('Invalid measurement')
12
- const median = values => { const sorted = [...values].sort((a,b) => a-b); const n = sorted.length; return n ? (sorted[Math.floor((n-1)/2)] + sorted[Math.floor(n/2)]) / 2 : null }
13
- const groups = [...new Set(rows.map(r => `${r.fixture}:${r.arm}`))].map(key => {
14
- const runs = rows.filter(r => `${r.fixture}:${r.arm}` === key)
15
- if (new Set(runs.map(r => `${r.model}:${r.effort}:${r.routing}`)).size !== 1) throw new Error('Incompatible execution policies in one comparison group')
16
- return { fixture: runs[0].fixture, arm: runs[0].arm, runs: runs.length, accepted: runs.filter(r => r.accepted).length,
17
- medianWallMs: median(runs.map(r => r.wallMs)),
18
- medianFreshInputTokens: runs.every(r => r.freshInputTokens !== null) ? median(runs.map(r => r.freshInputTokens)) : null,
19
- medianOutputTokens: runs.every(r => Number.isFinite(r.usage?.outputTokens)) ? median(runs.map(r => r.usage.outputTokens)) : null,
20
- medianAllInputTokens: runs.every(r => Number.isFinite(r.usage?.inputTokens)) ? median(runs.map(r => r.usage.inputTokens)) : null }
21
- })
22
- console.log(JSON.stringify({ auditedVersion: '1.19.0', auditedCommit: '05fac3db1a51412563005e07f80e6e5a9aab8e06', models: [...new Set(rows.map(r => r.model))], effort: 'low', groups, runs: rows, limitations: [...new Set(sources.flatMap(s => s.limits))], pricesMeasured: false, cpuMeasured: false, memoryMeasured: false }, null, 2))
69
+ const groups = []
70
+ for (const comparison of comparisons) {
71
+ const rows = cohorts.get(comparison.id).runs
72
+ runs.push(...rows.map(row => ({ ...row, comparisonId: comparison.id, evidenceStatus: comparison.status })))
73
+ if (comparison.status === 'verified') {
74
+ groups.push(...armSummaries(rows).map(group => ({ comparisonId: comparison.id, ...group })))
75
+ }
76
+ }
77
+ const verifiedSources = comparisons.filter(comparison => comparison.status === 'verified').flatMap(comparison => comparison.provenance)
78
+ const unique = values => [...new Set(values)]
79
+ const versions = unique(verifiedSources.map(source => source.version))
80
+ const commits = unique(verifiedSources.map(source => source.commit))
81
+ console.log(JSON.stringify({
82
+ schemaVersion: 2,
83
+ auditedVersion: versions.length === 1 ? versions[0] : null,
84
+ auditedCommit: commits.length === 1 ? commits[0] : null,
85
+ requestedModels: unique(runs.map(row => row.context?.execution?.requestedModel ?? row.model).filter(Boolean)),
86
+ reportedModels: unique(runs.flatMap(row => row.context?.execution?.actualModels ?? [])),
87
+ comparisons, groups, legacyReports, runs,
88
+ limitations: unique([
89
+ ...sources.flatMap(source => source.limits ?? source.limitations ?? []),
90
+ 'Verified means the recorded manifest conditions match; it does not certify model behavior, host isolation, or statistical significance',
91
+ 'Diagnostic summaries include unsuccessful or unverified runs and must not be used as accepted performance comparisons',
92
+ 'Missing or partial token telemetry remains unknown; no USD price is inferred',
93
+ ]),
94
+ pricesMeasured: false, cpuMeasured: false, memoryMeasured: false,
95
+ }, null, 2))
@@ -1,71 +1,194 @@
1
- // Controlled live comparison. Run after npm run build.
2
- // node bench/compare-codex.mjs --repeats=2 --fixture=routing-queue
3
- import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync } from 'node:fs'
4
- import { createHash } from 'node:crypto'
1
+ // Controlled live workflow comparison. Run only when authenticated runs are authorized.
2
+ // Run after npm run build. See docs/BENCHMARK-MANIFEST.md for scope and limits.
3
+ import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, lstatSync, existsSync } from 'node:fs'
4
+ import { createHash, randomUUID } from 'node:crypto'
5
5
  import { spawn, spawnSync } from 'node:child_process'
6
+ import { arch, cpus, hostname, release, tmpdir, totalmem } from 'node:os'
6
7
  import { dirname, join, resolve } from 'node:path'
7
8
  import { fileURLToPath } from 'node:url'
8
9
  import { parse, stringify } from 'yaml'
9
10
  import { buildProviderInvocation, startProviderProcess } from '../dist/agents/providers.js'
11
+ import { readEvents } from '../dist/observability/events.js'
12
+ import { stableComparisonJson } from './result-schema.mjs'
13
+
10
14
  const repo = dirname(dirname(fileURLToPath(import.meta.url)))
11
15
  const version = JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version
12
- const opts = Object.fromEntries(process.argv.slice(2).map(x => x.replace(/^--/, '').split('=')))
16
+ const opts = Object.fromEntries(process.argv.slice(2).map(value => value.replace(/^--/, '').split('=')))
13
17
  const fixture = opts.fixture ?? 'routing-queue'
14
18
  if (!['routing-queue', 'string-kit', 'independent-utils'].includes(fixture)) throw new Error('Unsupported fixture')
15
19
  const repeats = Number(opts.repeats ?? 2)
16
20
  if (!Number.isInteger(repeats) || repeats < 1 || repeats > 10) throw new Error('repeats must be 1..10')
17
21
  const model = opts.model ?? 'gpt-6.1-sol'
18
- const root = resolve(opts.root ?? `G:/NN-Developed/Yoke-Testground/b${Date.now().toString(36)}`)
22
+ const effort = opts.effort ?? 'low'
23
+ const root = resolve(opts.root ?? join(tmpdir(), 'ykb-' + Date.now().toString(36)))
19
24
  mkdirSync(root, { recursive: true })
20
25
  const seed = join(repo, 'bench', 'fixtures', fixture)
21
- const stories = parse(readFileSync(join(seed, '.yoke/prd.yaml'), 'utf8'))
22
- const originalTests = readdirSync(join(seed, 'tests')).filter(f => f.endsWith('.test.mjs')).sort()
23
- const protectedFiles = new Map(['bench-verify.mjs', ...originalTests.map(f => join('tests', f))].map(file => [file, readFileSync(join(seed, file))]))
24
- const acceptanceDigest = createHash('sha256').update(Buffer.concat([...protectedFiles].flatMap(([file, bytes]) => [Buffer.from(file + '\0'), bytes]))).digest('hex')
25
- const prompt = 'Implement all requirements below. Run node bench-verify.mjs. Preserve tests and the verification script. Do not commit.\n' + stories.map(s => `${s.id}: ${s.title}\n${s.acceptance.join('\n')}`).join('\n\n')
26
+ const prdInput = readFileSync(join(seed, '.yoke/prd.yaml'), 'utf8')
27
+ const stories = parse(prdInput)
28
+ const originalTests = readdirSync(join(seed, 'tests')).filter(file => file.endsWith('.test.mjs')).sort()
29
+ const protectedFiles = new Map(['bench-verify.mjs', ...originalTests.map(file => join('tests', file))].map(file => [file, readFileSync(join(seed, file))]))
30
+ const digest = value => createHash('sha256').update(value).digest('hex')
31
+ const acceptanceDigest = digest(Buffer.concat([...protectedFiles].flatMap(([file, bytes]) => [Buffer.from(file.replaceAll('\\', '/') + '\0'), bytes])))
32
+ const prompt = 'Implement all requirements below. Run node bench-verify.mjs. Preserve tests and the verification script. Do not commit.\n' + stories.map(story => story.id + ': ' + story.title + '\n' + story.acceptance.join('\n')).join('\n\n')
26
33
  const results = []
27
34
  const requestedArms = opts.arms?.split(',')
28
35
  if (requestedArms && (!requestedArms.length || requestedArms.some(arm => !['codex', 'yoke-serial', 'yoke-parallel'].includes(arm)))) throw new Error('arms must name codex,yoke-serial,yoke-parallel')
36
+ const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
37
+ const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
38
+ if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
39
+ const manifest = {
40
+ schemaVersion: 1, id: randomUUID(), kind: 'workflow', arms: selectedArms, repeats,
41
+ allowedDifferences: [
42
+ { field: 'execution.workflow', reason: 'One direct provider session versus the normal Yoke story workflow' },
43
+ { field: 'execution.parallel', reason: 'The explicitly selected parallel arm uses three Yoke workers' },
44
+ { field: 'execution.promptDigest', reason: 'Workflow inputs differ while the fixture requirements and acceptance remain identical' },
45
+ { field: 'execution.promptScope', reason: 'Direct provider prompt versus PRD/configuration inputs expanded by Yoke' },
46
+ { field: 'startup.ignoreRules', reason: 'Preserve the documented direct-arm ignore-rules flag; Yoke retains normal project rule handling' },
47
+ { field: 'startup.commitPolicy', reason: 'Direct arm is instructed not to commit; Yoke performs its normal story commits' },
48
+ { field: 'startup.isolation', reason: 'Yoke uses its normal isolated worktrees inside a fresh fixture' },
49
+ { field: 'startup.timeoutPolicy', reason: 'Direct session deadline versus bounded Yoke story dispatches' },
50
+ { field: 'startup.userStatePolicy', reason: 'Yoke runtime registry is isolated using LOCALAPPDATA; direct CLI keeps inherited state' },
51
+ ],
52
+ }
53
+ const limits = [
54
+ 'Small fixture sample; not representative of production projects',
55
+ 'Workflow comparison includes declared prompt, commit, isolation, timeout and rule-policy differences',
56
+ 'CPU and RAM not measured; host background load is uncontrolled',
57
+ 'Routing disabled; fixed requested model and effort; missing reported model identity prevents a verified comparison',
58
+ 'Immutable tests replayed after execution; visible to agents, not hidden tests',
59
+ 'User config disabled; installed plugins and discovered skills are not fully audited',
60
+ 'Prompt digests bind submitted workflow inputs, not provider system prompts or every dynamically expanded Yoke prompt',
61
+ 'Wall time starts after fixture setup and ends at runner exit; independent acceptance time is recorded separately',
62
+ ]
63
+ function hashPaths(base, paths) {
64
+ const hash = createHash('sha256')
65
+ function visit(relative) {
66
+ const path = join(base, relative)
67
+ const label = relative.replaceAll('\\', '/')
68
+ if (!existsSync(path)) { hash.update('missing:' + label + '\0'); return }
69
+ const stat = lstatSync(path)
70
+ if (stat.isSymbolicLink()) throw new Error('Cannot certify linked benchmark input: ' + label)
71
+ if (stat.isDirectory()) {
72
+ hash.update('directory:' + label + '\0')
73
+ for (const name of readdirSync(path).sort()) visit(join(relative, name))
74
+ } else if (stat.isFile()) {
75
+ const bytes = readFileSync(path)
76
+ hash.update(Buffer.byteLength(label) + ':' + label + ':' + bytes.length + ':')
77
+ hash.update(bytes)
78
+ } else throw new Error('Unsupported benchmark input: ' + label)
79
+ }
80
+ for (const path of [...paths].sort()) visit(path)
81
+ return hash.digest('hex')
82
+ }
83
+ function git(args) {
84
+ const result = spawnSync('git', args, { cwd: repo, encoding: 'utf8' })
85
+ return result.status === 0 ? result.stdout.trim() : null
86
+ }
87
+ function sourceSnapshot() {
88
+ const gitRoot = git(['rev-parse', '--show-toplevel'])
89
+ const sameRoot = gitRoot && (process.platform === 'win32' ? resolve(gitRoot).toLowerCase() === repo.toLowerCase() : resolve(gitRoot) === repo)
90
+ const status = sameRoot ? git(['status', '--porcelain', '--untracked-files=normal']) : null
91
+ return {
92
+ version: JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version,
93
+ commit: sameRoot ? git(['rev-parse', 'HEAD']) : null,
94
+ dirty: status === null ? null : status.length > 0,
95
+ // Hash actual runtime artifacts even on a clean checkout: dist may be stale or untracked.
96
+ buildDigest: hashPaths(repo, ['dist', 'canon', 'agents', 'hooks', 'package.json', 'package-lock.json', 'bench/compare-codex.mjs', 'bench/result-schema.mjs']),
97
+ }
98
+ }
99
+ function environmentSnapshot() {
100
+ const provider = spawnSync('codex', ['--version'], { encoding: 'utf8', timeout: 20_000, shell: process.platform === 'win32' })
101
+ return {
102
+ platform: process.platform, arch: arch(), nodeVersion: process.version,
103
+ providerVersion: provider.status === 0 ? provider.stdout.trim() || null : null,
104
+ hostDigest: digest(stableComparisonJson({ hostname: hostname(), platform: process.platform, release: release(), arch: arch(), cpus: cpus().map(cpu => cpu.model), memory: totalmem() })),
105
+ hostLoad: 'uncontrolled', skillsPlugins: 'not-audited',
106
+ }
107
+ }
108
+ function reportedModelsFromEvents(dir) {
109
+ const records = readEvents(dir, 1000).filter(event => event.type === 'tokens')
110
+ const calls = records.flatMap(event => Array.isArray(event.data?.calls) && event.data.calls.length ? event.data.calls : [event.data ?? {}])
111
+ const models = calls.map(call => call.actualModel ?? call.model)
112
+ return models.length && models.every(value => typeof value === 'string' && value) ? [...new Set(models)].sort() : null
113
+ }
114
+ const fixtureContext = {
115
+ id: fixture, seedDigest: hashPaths(seed, ['.']),
116
+ acceptanceDigest, requirementsDigest: digest(stableComparisonJson(stories)),
117
+ }
29
118
  for (let repeat = 0; repeat < repeats; repeat++) {
30
- const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
31
- const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
32
- if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
33
119
  const arms = repeat % 2 ? [...selectedArms].reverse() : selectedArms
34
120
  for (const arm of arms) {
35
- const dir = join(root, `${repeat + 1}-${arm}`)
36
- mkdirSync(dir); cpSync(seed, dir, { recursive: true })
121
+ const before = sourceSnapshot()
122
+ const environment = environmentSnapshot()
123
+ const dir = join(root, String(repeat + 1) + '-' + arm)
124
+ mkdirSync(dir)
125
+ cpSync(seed, dir, { recursive: true })
37
126
  const parallel = arm === 'yoke-parallel' ? 3 : 1
38
- writeFileSync(join(dir, '.yoke/config.yaml'), stringify({ canonVersion: '1.2.0', agents: ['codex'], loop: { enabled: true, parallel, isolate: true }, runner: { agent: 'codex', model, reasoningEffort: 'low', bare: true }, routing: { enabled: false }, verify: { command: 'node bench-verify.mjs' } }))
127
+ const configInput = stringify({ canonVersion: '1.2.0', agents: ['codex'], loop: { enabled: true, parallel, isolate: true }, runner: { agent: 'codex', model, reasoningEffort: effort, bare: true }, routing: { enabled: false }, verify: { command: 'node bench-verify.mjs' } })
128
+ writeFileSync(join(dir, '.yoke/config.yaml'), configInput)
39
129
  for (const args of [['init', '-q'], ['add', '-A'], ['-c', 'user.name=benchmark', '-c', 'user.email=benchmark@yoke.local', 'commit', '-qm', 'Immutable fixture seed']]) {
40
- const r = spawnSync('git', args, { cwd: dir, encoding: 'utf8' }); if (r.status !== 0) throw new Error(r.stderr)
130
+ const result = spawnSync('git', args, { cwd: dir, encoding: 'utf8' })
131
+ if (result.status !== 0) throw new Error(result.stderr)
41
132
  }
42
- console.log(JSON.stringify({ type: 'start', fixture, repeat: repeat + 1, arm, model, dir }))
43
- const start = Date.now(); let outcome; let events = []; let usage
133
+ console.log(JSON.stringify({ type: 'start', comparisonId: manifest.id, fixture, repeat: repeat + 1, arm, model, effort, dir }))
134
+ const start = Date.now()
135
+ let outcome, usage, actualModels = null
136
+ let events = []
44
137
  if (arm === 'codex') {
45
- const inv = buildProviderInvocation('codex', prompt, dir, 'unsafe', { model, reasoningEffort: 'low', bare: true, nativeMultiAgent: false })
46
- inv.args.push('--ignore-rules')
47
- const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), 600_000)
138
+ const invocation = buildProviderInvocation('codex', prompt, dir, 'unsafe', { model, reasoningEffort: effort, bare: true, nativeMultiAgent: false })
139
+ // Historical workflow policy is retained and explicitly declared, not silently equalized.
140
+ invocation.args.push('--ignore-rules')
141
+ const controller = new AbortController()
142
+ const timer = setTimeout(() => controller.abort(), 600_000)
48
143
  try {
49
- const r = await startProviderProcess('codex', inv, { signal: controller.signal, idleTimeoutMs: 180_000 }).completion
50
- outcome = r.kind; usage = r.telemetry.tokens
51
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.log`), JSON.stringify(r, null, 2))
144
+ const result = await startProviderProcess('codex', invocation, { signal: controller.signal, idleTimeoutMs: 180_000 }).completion
145
+ outcome = result.kind
146
+ usage = result.telemetry.tokens
147
+ const reported = result.telemetry.reportedModels ?? (usage?.model ? [usage.model] : [])
148
+ actualModels = reported.length ? [...new Set(reported)].sort() : null
149
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), JSON.stringify(result, null, 2))
52
150
  } finally { clearTimeout(timer) }
53
151
  } else {
54
- const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing', `--parallel=${parallel}`, '--max=3', '--timeout=10', '--unsafe'], { cwd: dir, env: { ...process.env, LOCALAPPDATA: join(root, 'state') }, stdio: ['ignore', 'pipe', 'pipe'] })
55
- let output = '', errors = ''; child.stdout.on('data', d => { output += d }); child.stderr.on('data', d => { errors += d })
56
- outcome = await new Promise((res, rej) => { child.on('error', rej); child.on('close', res) })
152
+ const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing', '--parallel=' + parallel, '--max=3', '--timeout=10', '--unsafe'], { cwd: dir, env: { ...process.env, LOCALAPPDATA: join(root, 'state') }, stdio: ['ignore', 'pipe', 'pipe'] })
153
+ let output = '', errors = ''
154
+ child.stdout.on('data', data => { output += data })
155
+ child.stderr.on('data', data => { errors += data })
156
+ outcome = await new Promise((resolveExit, reject) => { child.on('error', reject); child.on('close', resolveExit) })
57
157
  events = output.split('\n').flatMap(line => { try { return [JSON.parse(line)] } catch { return [] } })
58
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.log`), output + '\nSTDERR\n' + errors)
59
- usage = events.filter(e => e.type === 'status' && e.tokens).at(-1)?.tokens
158
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), output + '\nSTDERR\n' + errors)
159
+ usage = events.filter(event => event.type === 'status' && event.tokens).at(-1)?.tokens
160
+ actualModels = reportedModelsFromEvents(dir)
60
161
  }
61
162
  const wallMs = Date.now() - start
62
163
  // Replay immutable originals after the agent exits. Agent-written tests never determine quality.
63
164
  mkdirSync(join(dir, 'tests'), { recursive: true })
64
165
  for (const [file, bytes] of protectedFiles) writeFileSync(join(dir, file), bytes)
65
- const testFiles = originalTests.map(f => join('tests', f))
66
- const checkStart = Date.now(); const check = spawnSync(process.execPath, ['--test', ...testFiles], { cwd: dir, encoding: 'utf8', timeout: 60_000 })
67
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.acceptance.log`), check.stdout + check.stderr)
68
- const result = { fixture, repeat: repeat + 1, arm, model, effort: 'low', routing: false, nativeMultiAgent: false, parallel, acceptanceDigest, wallMs, acceptanceMs: Date.now() - checkStart, accepted: check.status === 0, outcome, usage: usage ?? null, events: events.length, dir }
69
- results.push(result); writeFileSync(join(root, 'results.json'), JSON.stringify({ version, root, results, limits: ['Small fixture sample; not representative of production projects', 'CPU and RAM not measured', 'Routing disabled; fixed model and effort', 'Immutable tests replayed after execution; not hidden from agents', 'User config disabled; not an audit of every installed plugin'] }, null, 2)); console.log(JSON.stringify({ type: 'result', ...result }))
166
+ const checkStart = Date.now()
167
+ const check = spawnSync(process.execPath, ['--test', ...originalTests.map(file => join('tests', file))], { cwd: dir, encoding: 'utf8', timeout: 60_000 })
168
+ const acceptanceMs = Date.now() - checkStart
169
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.acceptance.log'), check.stdout + check.stderr)
170
+ const after = sourceSnapshot()
171
+ const context = {
172
+ source: { ...before, stable: stableComparisonJson(before) === stableComparisonJson(after) && hashPaths(seed, ['.']) === fixtureContext.seedDigest },
173
+ fixture: fixtureContext,
174
+ execution: {
175
+ provider: 'codex', requestedModel: model, actualModels, effort, routing: false, nativeMultiAgent: false, nativeGoal: false, parallel,
176
+ workflow: arm === 'codex' ? 'direct-provider-session' : 'yoke-story-loop',
177
+ promptDigest: digest(arm === 'codex' ? prompt : stableComparisonJson({ prdInput, configInput })),
178
+ promptScope: arm === 'codex' ? 'provider-prompt' : 'workflow-input-bundle',
179
+ },
180
+ startup: {
181
+ permissionProfile: 'unsafe', bare: true, ignoreRules: arm === 'codex',
182
+ commitPolicy: arm === 'codex' ? 'instructed-no-commit' : 'normal-story-commits',
183
+ isolation: arm === 'codex' ? 'fresh-fixture' : 'fresh-fixture-and-story-worktrees',
184
+ timeoutPolicy: arm === 'codex' ? '600000ms session; 180000ms idle' : 'three dispatches; 600000ms per runner',
185
+ userStatePolicy: arm === 'codex' ? 'inherited' : 'isolated-LOCALAPPDATA',
186
+ },
187
+ environment,
188
+ }
189
+ const result = { fixture, repeat: repeat + 1, arm, model, effort, routing: false, nativeMultiAgent: false, parallel, acceptanceDigest, context, wallMs, acceptanceMs, accepted: check.status === 0, outcome, usage: usage ?? null, events: events.length, dir }
190
+ results.push(result)
191
+ writeFileSync(join(root, 'results.json'), JSON.stringify({ schemaVersion: 2, version, manifest, root, results, limits }, null, 2))
192
+ console.log(JSON.stringify({ type: 'result', comparisonId: manifest.id, ...result }))
70
193
  }
71
194
  }
@@ -10,3 +10,135 @@ export function validateResult(result) {
10
10
  if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
11
  return result
12
12
  }
13
+
14
+ // Comparison manifests are separate from the historical individual-run schema above.
15
+ // A "verified" manifest describes comparable recorded conditions, not a causal or
16
+ // statistically conclusive performance result.
17
+ const record = value => value !== null && typeof value === 'object' && !Array.isArray(value)
18
+ const nonempty = value => typeof value === 'string' && value.trim().length > 0
19
+ const boolean = value => typeof value === 'boolean'
20
+ const sha256 = value => typeof value === 'string' && /^[a-f0-9]{64}$/.test(value)
21
+ const positiveInteger = value => Number.isSafeInteger(value) && value > 0
22
+
23
+ export const comparisonContextFields = {
24
+ 'source.version': nonempty,
25
+ 'source.commit': value => typeof value === 'string' && /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/.test(value),
26
+ 'source.dirty': boolean,
27
+ 'source.buildDigest': sha256,
28
+ 'source.stable': value => value === true,
29
+ 'fixture.id': nonempty,
30
+ 'fixture.seedDigest': sha256,
31
+ 'fixture.acceptanceDigest': sha256,
32
+ 'fixture.requirementsDigest': sha256,
33
+ 'execution.provider': nonempty,
34
+ 'execution.requestedModel': nonempty,
35
+ 'execution.actualModels': value => Array.isArray(value) && value.length > 0 && value.every(nonempty) && new Set(value).size === value.length,
36
+ 'execution.effort': nonempty,
37
+ 'execution.routing': boolean,
38
+ 'execution.nativeMultiAgent': boolean,
39
+ 'execution.nativeGoal': boolean,
40
+ 'execution.parallel': positiveInteger,
41
+ 'execution.workflow': nonempty,
42
+ 'execution.promptDigest': sha256,
43
+ 'execution.promptScope': value => ['provider-prompt', 'workflow-input-bundle'].includes(value),
44
+ 'startup.permissionProfile': value => ['safe', 'read-only', 'unsafe'].includes(value),
45
+ 'startup.bare': boolean,
46
+ 'startup.ignoreRules': boolean,
47
+ 'startup.commitPolicy': nonempty,
48
+ 'startup.isolation': nonempty,
49
+ 'startup.timeoutPolicy': nonempty,
50
+ 'startup.userStatePolicy': nonempty,
51
+ 'environment.platform': nonempty,
52
+ 'environment.arch': nonempty,
53
+ 'environment.nodeVersion': nonempty,
54
+ 'environment.providerVersion': nonempty,
55
+ 'environment.hostDigest': sha256,
56
+ 'environment.hostLoad': value => ['uncontrolled', 'isolated'].includes(value),
57
+ 'environment.skillsPlugins': value => ['not-audited', 'audited'].includes(value),
58
+ }
59
+
60
+ export const allowedComparisonVariables = Object.keys(comparisonContextFields)
61
+ .filter(field => !field.startsWith('fixture.') && field !== 'source.stable')
62
+
63
+ export function comparisonValue(context, field) {
64
+ return field.split('.').reduce((value, key) => record(value) ? value[key] : undefined, context)
65
+ }
66
+
67
+ export function stableComparisonJson(value) {
68
+ if (Array.isArray(value)) return '[' + value.map(stableComparisonJson).join(',') + ']'
69
+ if (record(value)) return '{' + Object.keys(value).sort().map(key => JSON.stringify(key) + ':' + stableComparisonJson(value[key])).join(',') + '}'
70
+ return JSON.stringify(value)
71
+ }
72
+
73
+ export function comparisonManifestIssues(manifest) {
74
+ if (!record(manifest)) return ['Missing versioned comparison manifest']
75
+ const issues = []
76
+ if (manifest.schemaVersion !== 1) issues.push('Unsupported comparison manifest version')
77
+ if (!nonempty(manifest.id)) issues.push('Missing comparison identity')
78
+ if (!['workflow', 'controlled'].includes(manifest.kind)) issues.push('Unknown comparison kind')
79
+ if (!Array.isArray(manifest.arms) || manifest.arms.length < 2 || !manifest.arms.every(nonempty) || new Set(manifest.arms).size !== manifest.arms.length) issues.push('Declare at least two distinct comparison arms')
80
+ if (!positiveInteger(manifest.repeats)) issues.push('Declare the planned number of paired repeats')
81
+ if (!Array.isArray(manifest.allowedDifferences)) issues.push('Declare allowed differences explicitly, including an empty list when appropriate')
82
+ else {
83
+ const fields = new Set()
84
+ for (const difference of manifest.allowedDifferences) {
85
+ if (!record(difference) || !allowedComparisonVariables.includes(difference.field) || !nonempty(difference.reason)) {
86
+ issues.push('Invalid comparison variable or missing rationale')
87
+ continue
88
+ }
89
+ if (fields.has(difference.field)) issues.push('Duplicate comparison variable: ' + difference.field)
90
+ fields.add(difference.field)
91
+ }
92
+ }
93
+ return issues
94
+ }
95
+
96
+ export function comparisonContextIssues(context) {
97
+ if (!record(context)) return ['Missing per-run comparison context']
98
+ const issues = []
99
+ for (const [field, validate] of Object.entries(comparisonContextFields)) {
100
+ if (!validate(comparisonValue(context, field))) issues.push('Missing, unknown or invalid condition: ' + field)
101
+ }
102
+ // Refuse silently ignored startup/source conditions introduced by a future writer.
103
+ for (const [section, fields] of Object.entries(context)) {
104
+ if (!record(fields)) { issues.push('Invalid context section: ' + section); continue }
105
+ for (const field of Object.keys(fields)) {
106
+ if (!(section + '.' + field in comparisonContextFields)) issues.push('Unknown comparison condition: ' + section + '.' + field)
107
+ }
108
+ }
109
+ return issues
110
+ }
111
+
112
+ export function assessComparison(manifests, runs) {
113
+ const manifest = manifests[0]
114
+ let reasons = manifests.flatMap(comparisonManifestIssues)
115
+ if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
116
+ if (manifests.some(value => stableComparisonJson(value) !== stableComparisonJson(manifest))) return { status: 'incompatible', reasons: ['Comparison manifests differ for the same identity'] }
117
+ reasons = runs.flatMap(run => comparisonContextIssues(run.context).map(issue => run.arm + ': ' + issue))
118
+ for (const run of runs) {
119
+ if (!manifest.arms.includes(run.arm) || !positiveInteger(run.repeat) || run.repeat > manifest.repeats) reasons.push('Unexpected arm or repeat')
120
+ if (run.fixture !== run.context?.fixture?.id) reasons.push('Fixture label disagrees with measured context')
121
+ if (typeof run.accepted !== 'boolean') reasons.push('Missing acceptance outcome')
122
+ }
123
+ if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
124
+ const keys = runs.map(run => run.arm + ':' + run.repeat)
125
+ if (new Set(keys).size !== keys.length) return { status: 'incompatible', reasons: ['Duplicate arm/repeat measurements'] }
126
+ const permitted = new Set(manifest.allowedDifferences.map(value => value.field))
127
+ for (const field of Object.keys(comparisonContextFields)) {
128
+ const valueOf = run => {
129
+ const value = comparisonValue(run.context, field)
130
+ return stableComparisonJson(field === 'execution.actualModels' ? [...value].sort() : value)
131
+ }
132
+ for (const arm of manifest.arms) {
133
+ if (new Set(runs.filter(run => run.arm === arm).map(valueOf)).size > 1) reasons.push('Conditions changed within arm ' + arm + ': ' + field)
134
+ }
135
+ if (!permitted.has(field) && new Set(runs.map(valueOf)).size > 1) reasons.push('Undeclared difference between arms: ' + field)
136
+ }
137
+ if (reasons.length) return { status: 'incompatible', reasons: [...new Set(reasons)] }
138
+ if (runs.length !== manifest.arms.length * manifest.repeats) return { status: 'incomplete', reasons: ['Not all declared paired runs are present'] }
139
+ if (runs.some(run => !run.accepted)) return { status: 'acceptance-failed', reasons: ['At least one arm failed immutable acceptance; failed elapsed time is not a speedup'] }
140
+ if (runs.some(run => run.usage?.measurementComplete === false || ['inputTokens', 'cachedInputTokens', 'outputTokens'].some(field => !Number.isFinite(run.usage?.[field]) || run.usage[field] < 0))) {
141
+ return { status: 'unverified', reasons: ['Incomplete token telemetry; retain diagnostic timings without an efficiency comparison'] }
142
+ }
143
+ return { status: 'verified', reasons: [] }
144
+ }
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.19.0
2
+ version: 1.22.0
3
3
  agents: [claude, codex, gemini, qwen, opencode, kilo, pi, hermes]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
@@ -39,10 +39,11 @@ export function createPiTelemetry() {
39
39
  if (complete) {
40
40
  // Optional totals must also cover every turn; omitted is not measured zero.
41
41
  const measured = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] === turns));
42
+ const partial = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] !== turns));
42
43
  return { usageAvailable: true, tokens: {
43
44
  ...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens,
44
45
  ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}),
45
- }, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
46
+ }, ...(Object.keys(partial).length ? { partialUsage: partial } : {}), ...(reportedModels.length > 1 ? { reportedModels } : {}) };
46
47
  }
47
48
  return { usageAvailable: false,
48
49
  ...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}),