@hecer/yoke 1.21.1 → 1.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +48 -0
  4. package/README.md +8 -1
  5. package/TODOS.md +6 -0
  6. package/bench/analyze-codex-comparison.mjs +90 -17
  7. package/bench/compare-codex.mjs +159 -36
  8. package/bench/result-schema.mjs +132 -0
  9. package/canon/manifest.yaml +1 -1
  10. package/canon/skills/visual-verification/SKILL.md +25 -2
  11. package/canon/tools/codex-rtk-hook.mjs +6 -16
  12. package/dist/agents/pi-telemetry.js +2 -1
  13. package/dist/agents/process-streams.js +12 -64
  14. package/dist/agents/provider-selection.js +12 -0
  15. package/dist/agents/telemetry.js +52 -52
  16. package/dist/change/inbox.js +8 -3
  17. package/dist/check/command.js +69 -17
  18. package/dist/check/delivery.js +121 -0
  19. package/dist/cli.js +91 -3
  20. package/dist/code-intelligence/adapters/mcp.js +1 -0
  21. package/dist/code-intelligence/budgets.js +138 -0
  22. package/dist/code-intelligence/contracts.js +2 -0
  23. package/dist/code-intelligence/coordinator.js +159 -85
  24. package/dist/code-intelligence/evidence.js +87 -34
  25. package/dist/code-intelligence/index.js +1 -0
  26. package/dist/code-intelligence/mcp-client.js +25 -6
  27. package/dist/code-intelligence/mcp-server.js +14 -11
  28. package/dist/code-intelligence/preflight.js +71 -0
  29. package/dist/dashboard/analytics.js +5 -3
  30. package/dist/goals/command.js +183 -53
  31. package/dist/goals/usage.js +87 -0
  32. package/dist/loop/cache-isolation.js +36 -0
  33. package/dist/loop/candidate-cleanup.js +47 -17
  34. package/dist/loop/candidates.js +17 -11
  35. package/dist/loop/dispatcher.js +89 -26
  36. package/dist/loop/failure.js +104 -0
  37. package/dist/loop/gate-snapshot.js +19 -0
  38. package/dist/loop/git.js +1 -1
  39. package/dist/loop/loop.js +124 -70
  40. package/dist/loop/parallel-adapters.js +57 -6
  41. package/dist/loop/parallel-command.js +49 -7
  42. package/dist/loop/proof-retention.js +70 -0
  43. package/dist/loop/recovery.js +23 -5
  44. package/dist/loop/reporter.js +22 -5
  45. package/dist/loop/run-command.js +101 -47
  46. package/dist/loop/runner.js +6 -5
  47. package/dist/loop/worker.js +152 -91
  48. package/dist/observability/history.js +2 -1
  49. package/dist/observability/invocation.js +42 -0
  50. package/dist/observability/local-report.js +120 -0
  51. package/dist/observability/usage.js +19 -0
  52. package/dist/prd/command.js +20 -7
  53. package/dist/prd/decompose.js +5 -2
  54. package/dist/retrofit/config.js +29 -2
  55. package/dist/retrofit/gitignore.js +12 -0
  56. package/dist/retrofit/planners/codex.js +20 -20
  57. package/dist/routing/attempts.js +241 -0
  58. package/dist/routing/capability.js +13 -9
  59. package/dist/routing/optimization.js +73 -0
  60. package/dist/routing/registry.js +7 -1
  61. package/dist/routing/router.js +282 -127
  62. package/dist/setup/command.js +8 -2
  63. package/dist/smoke/command.js +387 -85
  64. package/dist/update/check.js +1 -1
  65. package/docs/BENCHMARK-MANIFEST.md +131 -0
  66. package/docs/CODE-INTELLIGENCE.md +43 -1
  67. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  68. package/docs/DELIVERY-JOURNEYS.md +206 -0
  69. package/docs/ECONOMIC-ROUTING.md +180 -0
  70. package/docs/GOALS.md +61 -4
  71. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  72. package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
  73. package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
  74. package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
  75. package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
  76. package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
  77. package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
  78. package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
  79. package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
  80. package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
  81. package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
  82. package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
  83. package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
  84. package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
  85. package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
  86. package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
  87. package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
  88. package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
  89. package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
  90. package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
  91. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
  92. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
  93. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
  94. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
  95. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
  96. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
  97. package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
  98. package/docs/parallel-execution.md +37 -9
  99. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
  100. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
  101. package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
  102. package/gemini-extension.json +1 -1
  103. package/package.json +1 -1
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.21.1",
5
+ "version": "1.23.0",
6
6
  "description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi and Hermes: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.21.1",
3
+ "version": "1.23.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows for eight supported harnesses",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -1,5 +1,53 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.23.0 — 2026-10-04
4
+
5
+ ### Added
6
+ - Add read-only `yoke tools-preflight` and bounded local `yoke usage` reports. Separate configured tools from available capabilities, recorded usage from unknown coverage, and overlapping worker time from summed process durations.
7
+ - Retain immutable hashed runtime proofs before isolated candidate cleanup, with source, configuration and environment bindings. Preserve candidates when transfer or validation fails.
8
+ - Add optional `smoke.sourceIdentity: { path, sha256 }` configuration; production browser smoke now requires it and verifies served bytes before launch and after journeys.
9
+
10
+ ### Fixed
11
+ - Make setup/retrofit help read-only and reject unknown or contradictory mutating flags before dispatch while preserving recovery options.
12
+ - Ignore supervision runtime files, diagnose already tracked copies, render actual parallel workers and integration states, and stop reporting success merely when an integrator becomes idle.
13
+ - Preserve redacted browser launch causes instead of reporting every failure as a missing package. Keep OAuth, authorization and configured form secrets out of diagnostics.
14
+ - Migrate Yoke-owned Codex RTK hooks to the native protocol without deleting foreign hooks; honor configured Code Intelligence workspace identifiers and await MCP shutdown before temporary directory cleanup.
15
+ - Validate actual candidate writes and sibling collisions before and after integration gates. Reject shared writable worktree runtime caches and completion commands that change source or final assets.
16
+ - Preserve partial/missing usage markers through routing and include recorded serial implementation durations without counting parent/child usage twice.
17
+
18
+ ### Migration and validation limits
19
+ - Existing browser smoke projects must pin an app-specific served source path and its lowercase SHA-256. Update the pin intentionally after source changes; a static byte pin establishes response identity, not an independent deployment attestation.
20
+ - Shared package download stores remain allowed. Worktrees need their own `node_modules` root and writable `.vite-temp`, `.vite` and `.cache` directories. The guard covers these conventional roots, not arbitrary application-defined cache paths or concurrent changes after inspection.
21
+ - No dependency installer or retry policy was added: Yoke does not own one. Full-product model speed/cost and genuine npm/provider cold/warm cache comparisons remain unmeasured. The paired help experiment is a local regression check only.
22
+ - See the [analysis](docs/benchmarks/2026-10-04-efficiency/ANALYSE.md) and [validation record](docs/RELEASE-VALIDATION-1.23.0.md). Version metadata is prepared locally; publication requires independent review and release gates, followed by verified CI/npm evidence.
23
+
24
+ ## 1.22.0 — 2026-10-03
25
+
26
+ ### Added
27
+ - Add durable routing attempt reservations independent of optional analytics retention, per-call goal admission and persistent consumption accounting, including planning and interrupted calls.
28
+ - Add optional evidence-based economic ranking inside capability routing. Cost, speed and balanced objectives require comparable, complete execution evidence; configured capability floors and fallback limits remain authoritative.
29
+ - Add `yoke goal assess` to prepare a bound goal contract explicitly, and record versioned provenance for setup model profiles as configuration priors rather than measured price or capability claims.
30
+ - Add compact persistent failure signatures: repeated unchanged failures request diagnosis and then stop instead of exhausting repeated identical turns. Completion failures carry typed reasons into repair planning.
31
+ - Add optional multi-step browser journeys and structured proof reports. Project checks can bind declared artifacts and journeys to an acceptance digest, source fingerprint, runtime environment and individual requirement results.
32
+ - Add versioned direct-Codex benchmark manifests containing actual source/build, fixture, acceptance, model and startup-policy provenance. Only compatible verified pairs contribute to comparisons; legacy data remains diagnostic.
33
+
34
+ ### Fixed
35
+ - Retain incomplete parallel work across failure, pause and decision boundaries, including unsuccessful candidate races, and resume implementation separately from already prepared integration candidates.
36
+ - Honor ambiguity aborts consistently in serial and parallel execution. Reuse successful gates only on unchanged source, task and gate inputs; rerun them after relevant changes and integration.
37
+ - Prevent exhausted routing budgets from resetting after statistics eviction or failed registry writes. Preserve the cost of planning even when routing blocks before implementation.
38
+ - Preserve known partial provider usage without converting unknown calls to zero-cost work or discarding measured worker usage. Record PRD drafting, decomposition and change-planning/coverage calls on success and failure.
39
+ - Apply goal assessment policy and routing rules, reset provider-specific defaults when changing runners, and separate historic native Codex bindings from the currently executing provider.
40
+ - Correct Code Intelligence reference edges, report unverified index freshness and traversal coverage honestly, and enforce shared request deadlines and bounded UTF-8 responses including evidence metadata.
41
+ - Synchronize package, lockfile, Canon, provider manifests and README release metadata.
42
+
43
+ ### Migration and validation limits
44
+ - Existing configurations retain their routing ranking unless `routing.optimization` is added. New setup configurations select `balanced` with a minimum of 20 comparable samples per candidate; insufficient or incomplete evidence keeps the conservative order. Profile tiers and `costTier` are configuration priors, not live provider prices.
45
+ - Goal token ceilings are checked before each additional call. A provider reporting only at call completion can still exceed a ceiling within that call; the overrun is retained and prevents further budgeted dispatch. Unknown interrupted consumption requires an explicit budget decision.
46
+ - Retained work consumes disk space until successful integration or explicit reviewed cleanup. Candidate recovery resumes the first retained alternative; additional alternatives remain available for inspection and cleanup. Custom lifecycles without `retain` keep their existing cleanup policy. Interrupted attempts remain spent budget and are excluded from model-quality comparisons; recovery does not invent missing accounting history.
47
+ - Delivery declarations map project-authored commands to artifacts and journeys. Hashes compare pre-check and post-check artifact content; they do not detect changes reverted between snapshots or independently prove that a command exercised a particular device or deployment. Artifact snapshots are limited to 512 MiB per file, 1 GiB total and 10 seconds per snapshot. Build artifacts before `yoke check`; inspect and explicitly refresh protected acceptance after intentional manifest changes.
48
+ - Code Intelligence token budgets use a documented conservative UTF-8 estimate, not provider-measured token counts. A tiny budget can only return a bounded `BUDGET_EXCEEDED` error envelope; backend freshness stays unknown without independent index evidence.
49
+ - This release uses deterministic regression and packaging checks. No new paid model benchmark or general speed/cost improvement is claimed; see the [validation record](docs/RELEASE-VALIDATION-1.22.0.md).
50
+
3
51
  ## 1.21.1 — 2026-10-03
4
52
 
5
53
  ### Fixed
package/README.md CHANGED
@@ -11,7 +11,7 @@
11
11
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
12
12
  ![Node.js 20+](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
13
13
 
14
- <!-- yoke:version:start -->1.21.1<!-- yoke:version:end --> · <!-- yoke:tests:start -->1426<!-- yoke:tests:end --> test cases · <!-- yoke:skills:start -->34<!-- yoke:skills:end --> skills
14
+ <!-- yoke:version:start -->1.23.0<!-- yoke:version:end --> · <!-- yoke:tests:start -->1697<!-- yoke:tests:end --> test cases · <!-- yoke:skills:start -->34<!-- yoke:skills:end --> skills
15
15
 
16
16
  <!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi | Hermes<!-- yoke:agents:end -->
17
17
 
@@ -21,6 +21,8 @@ Yoke turns a goal into acceptance-tested stories, coordinates one or more coding
21
21
 
22
22
  ## How it works
23
23
 
24
+ Version 1.23 adds local `yoke usage [dir] --json` and `yoke tools-preflight [dir] --json` diagnostics. Browser smoke now requires `smoke.sourceIdentity` with an app-specific same-origin path and the lowercase SHA-256 of its served bytes. Pin a stable source response and refresh it deliberately after source changes; see [browser delivery configuration](docs/DELIVERY-JOURNEYS.md). Local version preparation and validation limits are recorded in [1.23 validation](docs/RELEASE-VALIDATION-1.23.0.md).
25
+
24
26
  ```mermaid
25
27
  flowchart LR
26
28
  A[Goal and acceptance criteria] --> B[PRD stories and dependencies]
@@ -41,6 +43,8 @@ Each story carries observable acceptance criteria and targeted test commands. Pa
41
43
  | **Verified completion** | Checks acceptance criteria, the project verify command, and any configured completion, review, quality, or browser gates before accepting work. |
42
44
  | **Parallel execution** | Schedules independent stories in isolated worktrees, respects dependencies and write scopes, and coordinates a shared worker limit across projects. |
43
45
  | **Durable goals** | Binds objectives to executable acceptance, shares worker capacity, records budgets, and optionally resumes native Codex goal threads. |
46
+ | **Measured routing** | Keeps hard attempt budgets independent of statistics, accounts for planning and reviews, and uses comparable complete evidence for optional economic model selection. |
47
+ | **Delivery evidence** | Records executable user journeys, acceptance results and stable declared build artifacts with their source and acceptance fingerprints. |
44
48
  | **Long-running autonomy** | Recovers from provider failures and blocked work. Optional exploration discovers new, repository-evidenced tasks after the planned backlog drains. |
45
49
  | **A choice of agents** | Uses the native CLI for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi, or Hermes. Install and authenticate the CLI you choose; Yoke does not bundle model runtimes or credentials. |
46
50
  | **Project visibility** | A local dashboard shows project status, **Workspace analytics**, **History**, and controls such as **Queue a change**, safe-boundary pause/resume, and operator notes. It runs with the local Yoke process. Dark and light themes are available. |
@@ -119,9 +123,12 @@ The dashboard is a local control room. It does not discover every process or run
119
123
  | [Continuous exploration](docs/CONTINUOUS-EXPLORATION.md) | Autonomous discovery, runtime limits, pause/resume, and stop detection |
120
124
  | [Project workflows](docs/VERIFIED-PROJECTS.md) | Setup, verification, goals, and execution defaults |
121
125
  | [Verified goals](docs/GOALS.md) | Acceptance binding, native Codex goals, budgets, and resource admission |
126
+ | [Economic routing](docs/ECONOMIC-ROUTING.md) | Durable attempt budgets, per-call costs, capability floors, and conservative model selection |
127
+ | [Delivery journeys](docs/DELIVERY-JOURNEYS.md) | Browser steps, project acceptance, artifact hashes, and proof limits |
122
128
  | [Code Intelligence](docs/CODE-INTELLIGENCE.md) | Optional Graphify, Serena, and Graft evidence providers |
123
129
  | [Dashboard](docs/DASHBOARD-EVOLUTION.md) | Project views, history, controls, and reporting boundaries |
124
130
  | [Benchmarks](bench/RESULTS.md) | Direct Codex comparison, routing studies, sample limits, and integration findings |
131
+ | [Benchmark manifests](docs/BENCHMARK-MANIFEST.md) | Comparable inputs, model identity, build provenance, and legacy-data limitations |
125
132
  | [Changelog](CHANGELOG.md) | Release features, behavior changes, and migration notes |
126
133
 
127
134
  ## Safety and limits
package/TODOS.md CHANGED
@@ -1,5 +1,11 @@
1
1
  # Yoke follow-up work
2
2
 
3
+ ## Deferred from the user-authorized 1.23.0 release
4
+
5
+ - E7: establish dependency-installer ownership before enforcing lockfile-keyed install reuse and offline/network failure classification; measure genuine cold/warm setup conditions.
6
+ - E8: emit explicit failure categories at production boundaries where the cause is known; preserve unknown observer/guardian/approval coverage.
7
+ - E9: collect paired full-product benchmarks and rerun the populated NEXUS layout regression with a platform-correct dependency tree. The copied macOS dependencies could not start Vite on Windows. No general speed/cost improvement is established yet.
8
+
3
9
  - Add provider-native output schemas when all three CLIs expose compatible stable APIs.
4
10
  - Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
5
11
  - Add signed provenance and attestations to npm and GitHub releases.
@@ -1,22 +1,95 @@
1
1
  // node bench/analyze-codex-comparison.mjs <results.json> ... > summary.json
2
2
  import { readFileSync } from 'node:fs'
3
+ import { assessComparison, stableComparisonJson } from './result-schema.mjs'
4
+
3
5
  const sources = process.argv.slice(2).map(path => JSON.parse(readFileSync(path, 'utf8')))
4
6
  if (!sources.length) throw new Error('Supply comparison results.json files')
5
- const rows = sources.flatMap(s => s.results).map(({ dir, ...row }) => ({
6
- ...row,
7
- freshInputTokens: Number.isFinite(row.usage?.inputTokens) && Number.isFinite(row.usage?.cachedInputTokens)
8
- ? row.usage.inputTokens - row.usage.cachedInputTokens : null,
7
+ const runs = []
8
+ const cohorts = new Map()
9
+ const legacyReports = []
10
+ const median = values => {
11
+ const sorted = [...values].sort((a, b) => a - b)
12
+ const n = sorted.length
13
+ return n ? (sorted[Math.floor((n - 1) / 2)] + sorted[Math.floor(n / 2)]) / 2 : null
14
+ }
15
+ const measured = value => Number.isFinite(value) && value >= 0
16
+ const metric = (rows, read) => rows.every(row => measured(read(row))) ? median(rows.map(read)) : null
17
+ function summarize(rows) {
18
+ return {
19
+ fixture: rows[0].fixture, arm: rows[0].arm, runs: rows.length,
20
+ accepted: rows.filter(row => row.accepted === true).length,
21
+ medianWallMs: median(rows.map(row => row.wallMs)),
22
+ medianElapsedToAcceptanceMs: metric(rows, row => measured(row.acceptanceMs) ? row.wallMs + row.acceptanceMs : null),
23
+ medianFreshInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.freshInputTokens),
24
+ medianOutputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.outputTokens),
25
+ medianAllInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.inputTokens),
26
+ }
27
+ }
28
+ function armSummaries(rows) {
29
+ return [...new Set(rows.map(row => row.arm))].map(arm => summarize(rows.filter(row => row.arm === arm)))
30
+ }
31
+ for (const [index, source] of sources.entries()) {
32
+ const input = source.results ?? source.runs
33
+ if (!Array.isArray(input) || !input.length) throw new Error('No completed runs in input ' + (index + 1))
34
+ const rows = input.map(({ dir, ...row }) => {
35
+ if (!measured(row.wallMs) || row.wallMs <= 0) throw new Error('Invalid measurement: wall time')
36
+ for (const field of ['inputTokens', 'cachedInputTokens', 'outputTokens']) {
37
+ if (row.usage?.[field] != null && !measured(row.usage[field])) throw new Error('Invalid measurement: ' + field)
38
+ }
39
+ if (row.acceptanceMs != null && !measured(row.acceptanceMs)) throw new Error('Invalid measurement: acceptance time')
40
+ const freshInputTokens = measured(row.usage?.inputTokens) && measured(row.usage?.cachedInputTokens)
41
+ ? row.usage.inputTokens - row.usage.cachedInputTokens : null
42
+ if (freshInputTokens !== null && freshInputTokens < 0) throw new Error('Invalid measurement: cache exceeds input')
43
+ return { ...row, freshInputTokens }
44
+ })
45
+ if (!source.manifest) {
46
+ legacyReports.push({
47
+ status: 'legacy/unverified', reportedVersion: source.version ?? source.auditedVersion ?? null,
48
+ reportedCommit: source.commit ?? source.auditedCommit ?? null, runs: rows.length,
49
+ reason: 'No versioned manifest; provenance and cross-arm conditions cannot be verified',
50
+ diagnosticSummaries: armSummaries(rows),
51
+ })
52
+ runs.push(...rows.map(row => ({ ...row, evidenceStatus: 'legacy/unverified' })))
53
+ continue
54
+ }
55
+ // A malformed identity remains its own unverified cohort, never joins another file.
56
+ const id = typeof source.manifest.id === 'string' && source.manifest.id ? source.manifest.id : 'invalid-input-' + index
57
+ const cohort = cohorts.get(id) ?? { id, manifests: [], runs: [] }
58
+ cohort.manifests.push(source.manifest)
59
+ cohort.runs.push(...rows)
60
+ cohorts.set(id, cohort)
61
+ }
62
+ const comparisons = [...cohorts.values()].map(cohort => ({
63
+ id: cohort.id,
64
+ ...assessComparison(cohort.manifests, cohort.runs),
65
+ manifest: cohort.manifests[0],
66
+ provenance: [...new Map(cohort.runs.map(row => [stableComparisonJson(row.context?.source), row.context?.source ?? null])).values()],
67
+ diagnosticSummaries: armSummaries(cohort.runs),
9
68
  }))
10
- if (!rows.length) throw new Error('No completed runs')
11
- for (const row of rows) if (!Number.isFinite(row.wallMs) || row.wallMs <= 0 || (row.freshInputTokens !== null && row.freshInputTokens < 0)) throw new Error('Invalid measurement')
12
- const median = values => { const sorted = [...values].sort((a,b) => a-b); const n = sorted.length; return n ? (sorted[Math.floor((n-1)/2)] + sorted[Math.floor(n/2)]) / 2 : null }
13
- const groups = [...new Set(rows.map(r => `${r.fixture}:${r.arm}`))].map(key => {
14
- const runs = rows.filter(r => `${r.fixture}:${r.arm}` === key)
15
- if (new Set(runs.map(r => `${r.model}:${r.effort}:${r.routing}`)).size !== 1) throw new Error('Incompatible execution policies in one comparison group')
16
- return { fixture: runs[0].fixture, arm: runs[0].arm, runs: runs.length, accepted: runs.filter(r => r.accepted).length,
17
- medianWallMs: median(runs.map(r => r.wallMs)),
18
- medianFreshInputTokens: runs.every(r => r.freshInputTokens !== null) ? median(runs.map(r => r.freshInputTokens)) : null,
19
- medianOutputTokens: runs.every(r => Number.isFinite(r.usage?.outputTokens)) ? median(runs.map(r => r.usage.outputTokens)) : null,
20
- medianAllInputTokens: runs.every(r => Number.isFinite(r.usage?.inputTokens)) ? median(runs.map(r => r.usage.inputTokens)) : null }
21
- })
22
- console.log(JSON.stringify({ auditedVersion: '1.19.0', auditedCommit: '05fac3db1a51412563005e07f80e6e5a9aab8e06', models: [...new Set(rows.map(r => r.model))], effort: 'low', groups, runs: rows, limitations: [...new Set(sources.flatMap(s => s.limits))], pricesMeasured: false, cpuMeasured: false, memoryMeasured: false }, null, 2))
69
+ const groups = []
70
+ for (const comparison of comparisons) {
71
+ const rows = cohorts.get(comparison.id).runs
72
+ runs.push(...rows.map(row => ({ ...row, comparisonId: comparison.id, evidenceStatus: comparison.status })))
73
+ if (comparison.status === 'verified') {
74
+ groups.push(...armSummaries(rows).map(group => ({ comparisonId: comparison.id, ...group })))
75
+ }
76
+ }
77
+ const verifiedSources = comparisons.filter(comparison => comparison.status === 'verified').flatMap(comparison => comparison.provenance)
78
+ const unique = values => [...new Set(values)]
79
+ const versions = unique(verifiedSources.map(source => source.version))
80
+ const commits = unique(verifiedSources.map(source => source.commit))
81
+ console.log(JSON.stringify({
82
+ schemaVersion: 2,
83
+ auditedVersion: versions.length === 1 ? versions[0] : null,
84
+ auditedCommit: commits.length === 1 ? commits[0] : null,
85
+ requestedModels: unique(runs.map(row => row.context?.execution?.requestedModel ?? row.model).filter(Boolean)),
86
+ reportedModels: unique(runs.flatMap(row => row.context?.execution?.actualModels ?? [])),
87
+ comparisons, groups, legacyReports, runs,
88
+ limitations: unique([
89
+ ...sources.flatMap(source => source.limits ?? source.limitations ?? []),
90
+ 'Verified means the recorded manifest conditions match; it does not certify model behavior, host isolation, or statistical significance',
91
+ 'Diagnostic summaries include unsuccessful or unverified runs and must not be used as accepted performance comparisons',
92
+ 'Missing or partial token telemetry remains unknown; no USD price is inferred',
93
+ ]),
94
+ pricesMeasured: false, cpuMeasured: false, memoryMeasured: false,
95
+ }, null, 2))
@@ -1,71 +1,194 @@
1
- // Controlled live comparison. Run after npm run build.
2
- // node bench/compare-codex.mjs --repeats=2 --fixture=routing-queue
3
- import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync } from 'node:fs'
4
- import { createHash } from 'node:crypto'
1
+ // Controlled live workflow comparison. Run only when authenticated runs are authorized.
2
+ // Run after npm run build. See docs/BENCHMARK-MANIFEST.md for scope and limits.
3
+ import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, lstatSync, existsSync } from 'node:fs'
4
+ import { createHash, randomUUID } from 'node:crypto'
5
5
  import { spawn, spawnSync } from 'node:child_process'
6
+ import { arch, cpus, hostname, release, tmpdir, totalmem } from 'node:os'
6
7
  import { dirname, join, resolve } from 'node:path'
7
8
  import { fileURLToPath } from 'node:url'
8
9
  import { parse, stringify } from 'yaml'
9
10
  import { buildProviderInvocation, startProviderProcess } from '../dist/agents/providers.js'
11
+ import { readEvents } from '../dist/observability/events.js'
12
+ import { stableComparisonJson } from './result-schema.mjs'
13
+
10
14
  const repo = dirname(dirname(fileURLToPath(import.meta.url)))
11
15
  const version = JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version
12
- const opts = Object.fromEntries(process.argv.slice(2).map(x => x.replace(/^--/, '').split('=')))
16
+ const opts = Object.fromEntries(process.argv.slice(2).map(value => value.replace(/^--/, '').split('=')))
13
17
  const fixture = opts.fixture ?? 'routing-queue'
14
18
  if (!['routing-queue', 'string-kit', 'independent-utils'].includes(fixture)) throw new Error('Unsupported fixture')
15
19
  const repeats = Number(opts.repeats ?? 2)
16
20
  if (!Number.isInteger(repeats) || repeats < 1 || repeats > 10) throw new Error('repeats must be 1..10')
17
21
  const model = opts.model ?? 'gpt-6.1-sol'
18
- const root = resolve(opts.root ?? `G:/NN-Developed/Yoke-Testground/b${Date.now().toString(36)}`)
22
+ const effort = opts.effort ?? 'low'
23
+ const root = resolve(opts.root ?? join(tmpdir(), 'ykb-' + Date.now().toString(36)))
19
24
  mkdirSync(root, { recursive: true })
20
25
  const seed = join(repo, 'bench', 'fixtures', fixture)
21
- const stories = parse(readFileSync(join(seed, '.yoke/prd.yaml'), 'utf8'))
22
- const originalTests = readdirSync(join(seed, 'tests')).filter(f => f.endsWith('.test.mjs')).sort()
23
- const protectedFiles = new Map(['bench-verify.mjs', ...originalTests.map(f => join('tests', f))].map(file => [file, readFileSync(join(seed, file))]))
24
- const acceptanceDigest = createHash('sha256').update(Buffer.concat([...protectedFiles].flatMap(([file, bytes]) => [Buffer.from(file + '\0'), bytes]))).digest('hex')
25
- const prompt = 'Implement all requirements below. Run node bench-verify.mjs. Preserve tests and the verification script. Do not commit.\n' + stories.map(s => `${s.id}: ${s.title}\n${s.acceptance.join('\n')}`).join('\n\n')
26
+ const prdInput = readFileSync(join(seed, '.yoke/prd.yaml'), 'utf8')
27
+ const stories = parse(prdInput)
28
+ const originalTests = readdirSync(join(seed, 'tests')).filter(file => file.endsWith('.test.mjs')).sort()
29
+ const protectedFiles = new Map(['bench-verify.mjs', ...originalTests.map(file => join('tests', file))].map(file => [file, readFileSync(join(seed, file))]))
30
+ const digest = value => createHash('sha256').update(value).digest('hex')
31
+ const acceptanceDigest = digest(Buffer.concat([...protectedFiles].flatMap(([file, bytes]) => [Buffer.from(file.replaceAll('\\', '/') + '\0'), bytes])))
32
+ const prompt = 'Implement all requirements below. Run node bench-verify.mjs. Preserve tests and the verification script. Do not commit.\n' + stories.map(story => story.id + ': ' + story.title + '\n' + story.acceptance.join('\n')).join('\n\n')
26
33
  const results = []
27
34
  const requestedArms = opts.arms?.split(',')
28
35
  if (requestedArms && (!requestedArms.length || requestedArms.some(arm => !['codex', 'yoke-serial', 'yoke-parallel'].includes(arm)))) throw new Error('arms must name codex,yoke-serial,yoke-parallel')
36
+ const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
37
+ const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
38
+ if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
39
+ const manifest = {
40
+ schemaVersion: 1, id: randomUUID(), kind: 'workflow', arms: selectedArms, repeats,
41
+ allowedDifferences: [
42
+ { field: 'execution.workflow', reason: 'One direct provider session versus the normal Yoke story workflow' },
43
+ { field: 'execution.parallel', reason: 'The explicitly selected parallel arm uses three Yoke workers' },
44
+ { field: 'execution.promptDigest', reason: 'Workflow inputs differ while the fixture requirements and acceptance remain identical' },
45
+ { field: 'execution.promptScope', reason: 'Direct provider prompt versus PRD/configuration inputs expanded by Yoke' },
46
+ { field: 'startup.ignoreRules', reason: 'Preserve the documented direct-arm ignore-rules flag; Yoke retains normal project rule handling' },
47
+ { field: 'startup.commitPolicy', reason: 'Direct arm is instructed not to commit; Yoke performs its normal story commits' },
48
+ { field: 'startup.isolation', reason: 'Yoke uses its normal isolated worktrees inside a fresh fixture' },
49
+ { field: 'startup.timeoutPolicy', reason: 'Direct session deadline versus bounded Yoke story dispatches' },
50
+ { field: 'startup.userStatePolicy', reason: 'Yoke runtime registry is isolated using LOCALAPPDATA; direct CLI keeps inherited state' },
51
+ ],
52
+ }
53
+ const limits = [
54
+ 'Small fixture sample; not representative of production projects',
55
+ 'Workflow comparison includes declared prompt, commit, isolation, timeout and rule-policy differences',
56
+ 'CPU and RAM not measured; host background load is uncontrolled',
57
+ 'Routing disabled; fixed requested model and effort; missing reported model identity prevents a verified comparison',
58
+ 'Immutable tests replayed after execution; visible to agents, not hidden tests',
59
+ 'User config disabled; installed plugins and discovered skills are not fully audited',
60
+ 'Prompt digests bind submitted workflow inputs, not provider system prompts or every dynamically expanded Yoke prompt',
61
+ 'Wall time starts after fixture setup and ends at runner exit; independent acceptance time is recorded separately',
62
+ ]
63
+ function hashPaths(base, paths) {
64
+ const hash = createHash('sha256')
65
+ function visit(relative) {
66
+ const path = join(base, relative)
67
+ const label = relative.replaceAll('\\', '/')
68
+ if (!existsSync(path)) { hash.update('missing:' + label + '\0'); return }
69
+ const stat = lstatSync(path)
70
+ if (stat.isSymbolicLink()) throw new Error('Cannot certify linked benchmark input: ' + label)
71
+ if (stat.isDirectory()) {
72
+ hash.update('directory:' + label + '\0')
73
+ for (const name of readdirSync(path).sort()) visit(join(relative, name))
74
+ } else if (stat.isFile()) {
75
+ const bytes = readFileSync(path)
76
+ hash.update(Buffer.byteLength(label) + ':' + label + ':' + bytes.length + ':')
77
+ hash.update(bytes)
78
+ } else throw new Error('Unsupported benchmark input: ' + label)
79
+ }
80
+ for (const path of [...paths].sort()) visit(path)
81
+ return hash.digest('hex')
82
+ }
83
+ function git(args) {
84
+ const result = spawnSync('git', args, { cwd: repo, encoding: 'utf8' })
85
+ return result.status === 0 ? result.stdout.trim() : null
86
+ }
87
+ function sourceSnapshot() {
88
+ const gitRoot = git(['rev-parse', '--show-toplevel'])
89
+ const sameRoot = gitRoot && (process.platform === 'win32' ? resolve(gitRoot).toLowerCase() === repo.toLowerCase() : resolve(gitRoot) === repo)
90
+ const status = sameRoot ? git(['status', '--porcelain', '--untracked-files=normal']) : null
91
+ return {
92
+ version: JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version,
93
+ commit: sameRoot ? git(['rev-parse', 'HEAD']) : null,
94
+ dirty: status === null ? null : status.length > 0,
95
+ // Hash actual runtime artifacts even on a clean checkout: dist may be stale or untracked.
96
+ buildDigest: hashPaths(repo, ['dist', 'canon', 'agents', 'hooks', 'package.json', 'package-lock.json', 'bench/compare-codex.mjs', 'bench/result-schema.mjs']),
97
+ }
98
+ }
99
+ function environmentSnapshot() {
100
+ const provider = spawnSync('codex', ['--version'], { encoding: 'utf8', timeout: 20_000, shell: process.platform === 'win32' })
101
+ return {
102
+ platform: process.platform, arch: arch(), nodeVersion: process.version,
103
+ providerVersion: provider.status === 0 ? provider.stdout.trim() || null : null,
104
+ hostDigest: digest(stableComparisonJson({ hostname: hostname(), platform: process.platform, release: release(), arch: arch(), cpus: cpus().map(cpu => cpu.model), memory: totalmem() })),
105
+ hostLoad: 'uncontrolled', skillsPlugins: 'not-audited',
106
+ }
107
+ }
108
+ function reportedModelsFromEvents(dir) {
109
+ const records = readEvents(dir, 1000).filter(event => event.type === 'tokens')
110
+ const calls = records.flatMap(event => Array.isArray(event.data?.calls) && event.data.calls.length ? event.data.calls : [event.data ?? {}])
111
+ const models = calls.map(call => call.actualModel ?? call.model)
112
+ return models.length && models.every(value => typeof value === 'string' && value) ? [...new Set(models)].sort() : null
113
+ }
114
+ const fixtureContext = {
115
+ id: fixture, seedDigest: hashPaths(seed, ['.']),
116
+ acceptanceDigest, requirementsDigest: digest(stableComparisonJson(stories)),
117
+ }
29
118
  for (let repeat = 0; repeat < repeats; repeat++) {
30
- const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
31
- const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
32
- if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
33
119
  const arms = repeat % 2 ? [...selectedArms].reverse() : selectedArms
34
120
  for (const arm of arms) {
35
- const dir = join(root, `${repeat + 1}-${arm}`)
36
- mkdirSync(dir); cpSync(seed, dir, { recursive: true })
121
+ const before = sourceSnapshot()
122
+ const environment = environmentSnapshot()
123
+ const dir = join(root, String(repeat + 1) + '-' + arm)
124
+ mkdirSync(dir)
125
+ cpSync(seed, dir, { recursive: true })
37
126
  const parallel = arm === 'yoke-parallel' ? 3 : 1
38
- writeFileSync(join(dir, '.yoke/config.yaml'), stringify({ canonVersion: '1.2.0', agents: ['codex'], loop: { enabled: true, parallel, isolate: true }, runner: { agent: 'codex', model, reasoningEffort: 'low', bare: true }, routing: { enabled: false }, verify: { command: 'node bench-verify.mjs' } }))
127
+ const configInput = stringify({ canonVersion: '1.2.0', agents: ['codex'], loop: { enabled: true, parallel, isolate: true }, runner: { agent: 'codex', model, reasoningEffort: effort, bare: true }, routing: { enabled: false }, verify: { command: 'node bench-verify.mjs' } })
128
+ writeFileSync(join(dir, '.yoke/config.yaml'), configInput)
39
129
  for (const args of [['init', '-q'], ['add', '-A'], ['-c', 'user.name=benchmark', '-c', 'user.email=benchmark@yoke.local', 'commit', '-qm', 'Immutable fixture seed']]) {
40
- const r = spawnSync('git', args, { cwd: dir, encoding: 'utf8' }); if (r.status !== 0) throw new Error(r.stderr)
130
+ const result = spawnSync('git', args, { cwd: dir, encoding: 'utf8' })
131
+ if (result.status !== 0) throw new Error(result.stderr)
41
132
  }
42
- console.log(JSON.stringify({ type: 'start', fixture, repeat: repeat + 1, arm, model, dir }))
43
- const start = Date.now(); let outcome; let events = []; let usage
133
+ console.log(JSON.stringify({ type: 'start', comparisonId: manifest.id, fixture, repeat: repeat + 1, arm, model, effort, dir }))
134
+ const start = Date.now()
135
+ let outcome, usage, actualModels = null
136
+ let events = []
44
137
  if (arm === 'codex') {
45
- const inv = buildProviderInvocation('codex', prompt, dir, 'unsafe', { model, reasoningEffort: 'low', bare: true, nativeMultiAgent: false })
46
- inv.args.push('--ignore-rules')
47
- const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), 600_000)
138
+ const invocation = buildProviderInvocation('codex', prompt, dir, 'unsafe', { model, reasoningEffort: effort, bare: true, nativeMultiAgent: false })
139
+ // Historical workflow policy is retained and explicitly declared, not silently equalized.
140
+ invocation.args.push('--ignore-rules')
141
+ const controller = new AbortController()
142
+ const timer = setTimeout(() => controller.abort(), 600_000)
48
143
  try {
49
- const r = await startProviderProcess('codex', inv, { signal: controller.signal, idleTimeoutMs: 180_000 }).completion
50
- outcome = r.kind; usage = r.telemetry.tokens
51
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.log`), JSON.stringify(r, null, 2))
144
+ const result = await startProviderProcess('codex', invocation, { signal: controller.signal, idleTimeoutMs: 180_000 }).completion
145
+ outcome = result.kind
146
+ usage = result.telemetry.tokens
147
+ const reported = result.telemetry.reportedModels ?? (usage?.model ? [usage.model] : [])
148
+ actualModels = reported.length ? [...new Set(reported)].sort() : null
149
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), JSON.stringify(result, null, 2))
52
150
  } finally { clearTimeout(timer) }
53
151
  } else {
54
- const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing', `--parallel=${parallel}`, '--max=3', '--timeout=10', '--unsafe'], { cwd: dir, env: { ...process.env, LOCALAPPDATA: join(root, 'state') }, stdio: ['ignore', 'pipe', 'pipe'] })
55
- let output = '', errors = ''; child.stdout.on('data', d => { output += d }); child.stderr.on('data', d => { errors += d })
56
- outcome = await new Promise((res, rej) => { child.on('error', rej); child.on('close', res) })
152
+ const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing', '--parallel=' + parallel, '--max=3', '--timeout=10', '--unsafe'], { cwd: dir, env: { ...process.env, LOCALAPPDATA: join(root, 'state') }, stdio: ['ignore', 'pipe', 'pipe'] })
153
+ let output = '', errors = ''
154
+ child.stdout.on('data', data => { output += data })
155
+ child.stderr.on('data', data => { errors += data })
156
+ outcome = await new Promise((resolveExit, reject) => { child.on('error', reject); child.on('close', resolveExit) })
57
157
  events = output.split('\n').flatMap(line => { try { return [JSON.parse(line)] } catch { return [] } })
58
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.log`), output + '\nSTDERR\n' + errors)
59
- usage = events.filter(e => e.type === 'status' && e.tokens).at(-1)?.tokens
158
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), output + '\nSTDERR\n' + errors)
159
+ usage = events.filter(event => event.type === 'status' && event.tokens).at(-1)?.tokens
160
+ actualModels = reportedModelsFromEvents(dir)
60
161
  }
61
162
  const wallMs = Date.now() - start
62
163
  // Replay immutable originals after the agent exits. Agent-written tests never determine quality.
63
164
  mkdirSync(join(dir, 'tests'), { recursive: true })
64
165
  for (const [file, bytes] of protectedFiles) writeFileSync(join(dir, file), bytes)
65
- const testFiles = originalTests.map(f => join('tests', f))
66
- const checkStart = Date.now(); const check = spawnSync(process.execPath, ['--test', ...testFiles], { cwd: dir, encoding: 'utf8', timeout: 60_000 })
67
- writeFileSync(join(root, `${fixture}-${repeat + 1}-${arm}.acceptance.log`), check.stdout + check.stderr)
68
- const result = { fixture, repeat: repeat + 1, arm, model, effort: 'low', routing: false, nativeMultiAgent: false, parallel, acceptanceDigest, wallMs, acceptanceMs: Date.now() - checkStart, accepted: check.status === 0, outcome, usage: usage ?? null, events: events.length, dir }
69
- results.push(result); writeFileSync(join(root, 'results.json'), JSON.stringify({ version, root, results, limits: ['Small fixture sample; not representative of production projects', 'CPU and RAM not measured', 'Routing disabled; fixed model and effort', 'Immutable tests replayed after execution; not hidden from agents', 'User config disabled; not an audit of every installed plugin'] }, null, 2)); console.log(JSON.stringify({ type: 'result', ...result }))
166
+ const checkStart = Date.now()
167
+ const check = spawnSync(process.execPath, ['--test', ...originalTests.map(file => join('tests', file))], { cwd: dir, encoding: 'utf8', timeout: 60_000 })
168
+ const acceptanceMs = Date.now() - checkStart
169
+ writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.acceptance.log'), check.stdout + check.stderr)
170
+ const after = sourceSnapshot()
171
+ const context = {
172
+ source: { ...before, stable: stableComparisonJson(before) === stableComparisonJson(after) && hashPaths(seed, ['.']) === fixtureContext.seedDigest },
173
+ fixture: fixtureContext,
174
+ execution: {
175
+ provider: 'codex', requestedModel: model, actualModels, effort, routing: false, nativeMultiAgent: false, nativeGoal: false, parallel,
176
+ workflow: arm === 'codex' ? 'direct-provider-session' : 'yoke-story-loop',
177
+ promptDigest: digest(arm === 'codex' ? prompt : stableComparisonJson({ prdInput, configInput })),
178
+ promptScope: arm === 'codex' ? 'provider-prompt' : 'workflow-input-bundle',
179
+ },
180
+ startup: {
181
+ permissionProfile: 'unsafe', bare: true, ignoreRules: arm === 'codex',
182
+ commitPolicy: arm === 'codex' ? 'instructed-no-commit' : 'normal-story-commits',
183
+ isolation: arm === 'codex' ? 'fresh-fixture' : 'fresh-fixture-and-story-worktrees',
184
+ timeoutPolicy: arm === 'codex' ? '600000ms session; 180000ms idle' : 'three dispatches; 600000ms per runner',
185
+ userStatePolicy: arm === 'codex' ? 'inherited' : 'isolated-LOCALAPPDATA',
186
+ },
187
+ environment,
188
+ }
189
+ const result = { fixture, repeat: repeat + 1, arm, model, effort, routing: false, nativeMultiAgent: false, parallel, acceptanceDigest, context, wallMs, acceptanceMs, accepted: check.status === 0, outcome, usage: usage ?? null, events: events.length, dir }
190
+ results.push(result)
191
+ writeFileSync(join(root, 'results.json'), JSON.stringify({ schemaVersion: 2, version, manifest, root, results, limits }, null, 2))
192
+ console.log(JSON.stringify({ type: 'result', comparisonId: manifest.id, ...result }))
70
193
  }
71
194
  }