@hecer/yoke 1.5.1 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. package/.claude-plugin/plugin.json +13 -13
  2. package/.codex-plugin/plugin.json +7 -7
  3. package/CHANGELOG.md +280 -259
  4. package/README.md +855 -834
  5. package/TODOS.md +5 -5
  6. package/agents/docs.toml +6 -6
  7. package/agents/implementer.toml +6 -6
  8. package/agents/reviewer.toml +6 -6
  9. package/agents/security.toml +6 -6
  10. package/bench/README.md +86 -86
  11. package/bench/RESULTS.md +35 -35
  12. package/bench/output-compaction.mjs +65 -65
  13. package/bench/result-schema.mjs +12 -12
  14. package/bench/results/claude-2026-07-27T18-03-26.json +50 -50
  15. package/bench/results/codex-unavailable-1785175418318.json +15 -15
  16. package/bench/results/gemini-2026-07-27T18-03-44.json +46 -46
  17. package/bench/run-matrix.mjs +26 -26
  18. package/bench/run.mjs +106 -106
  19. package/canon/AGENTS.md +30 -30
  20. package/canon/context/DECISIONS.md +4 -4
  21. package/canon/context/GLOSSARY.md +11 -0
  22. package/canon/context/KNOWLEDGE.md +4 -4
  23. package/canon/context/PROJECT.md +15 -15
  24. package/canon/loop/loop-spec.md +65 -65
  25. package/canon/loop/prd.schema.md +43 -43
  26. package/canon/manifest.yaml +59 -53
  27. package/canon/policy/gates.md +7 -7
  28. package/canon/policy/roles.md +9 -9
  29. package/canon/skills/ATTRIBUTION.md +99 -71
  30. package/canon/skills/authoring-prd/SKILL.md +58 -58
  31. package/canon/skills/brainstorming/SKILL.md +164 -164
  32. package/canon/skills/codebase-design/DEEPENING.md +15 -0
  33. package/canon/skills/codebase-design/DESIGN-IT-TWICE.md +12 -0
  34. package/canon/skills/codebase-design/SKILL.md +39 -0
  35. package/canon/skills/dispatching-parallel-agents/SKILL.md +182 -182
  36. package/canon/skills/document-release/SKILL.md +302 -297
  37. package/canon/skills/domain-modeling/ADR-FORMAT.md +19 -0
  38. package/canon/skills/domain-modeling/CONTEXT-FORMAT.md +39 -0
  39. package/canon/skills/domain-modeling/SKILL.md +35 -0
  40. package/canon/skills/executing-plans/SKILL.md +70 -70
  41. package/canon/skills/finishing-a-development-branch/SKILL.md +200 -200
  42. package/canon/skills/health/SKILL.md +177 -177
  43. package/canon/skills/maintaining-context/SKILL.md +34 -34
  44. package/canon/skills/minimal-code/SKILL.md +21 -21
  45. package/canon/skills/no-ai-slop/SKILL.md +103 -0
  46. package/canon/skills/no-ai-slop/eval.md +43 -0
  47. package/canon/skills/plan-ceo-review/SKILL.md +541 -541
  48. package/canon/skills/plan-eng-review/SKILL.md +362 -362
  49. package/canon/skills/receiving-code-review/SKILL.md +213 -213
  50. package/canon/skills/requesting-code-review/SKILL.md +105 -105
  51. package/canon/skills/resolving-merge-conflicts/SKILL.md +18 -0
  52. package/canon/skills/retro/SKILL.md +397 -397
  53. package/canon/skills/review/SKILL.md +246 -246
  54. package/canon/skills/ship/SKILL.md +691 -691
  55. package/canon/skills/subagent-driven-development/SKILL.md +277 -277
  56. package/canon/skills/systematic-debugging/SKILL.md +296 -296
  57. package/canon/skills/tdd/SKILL.md +371 -371
  58. package/canon/skills/unslop-ui/SKILL.md +34 -34
  59. package/canon/skills/using-git-worktrees/SKILL.md +218 -218
  60. package/canon/skills/verification-before-completion/SKILL.md +139 -139
  61. package/canon/skills/visual-verification/SKILL.md +54 -54
  62. package/canon/skills/workflow/SKILL.md +22 -22
  63. package/canon/skills/writing-for-agents/SKILL-MECHANICS.md +27 -0
  64. package/canon/skills/writing-for-agents/SKILL.md +42 -0
  65. package/canon/skills/writing-plans/SKILL.md +152 -152
  66. package/canon/skills/writing-skills/SKILL.md +655 -655
  67. package/canon/skills/yoke-retrofit/SKILL.md +26 -26
  68. package/canon/skills/yoke-workflow/SKILL.md +20 -20
  69. package/canon/tools/codex-rtk-hook.mjs +35 -35
  70. package/canon/tools/graphify.md +3 -3
  71. package/canon/tools/playwright-mcp.md +3 -3
  72. package/canon/tools/rtk.md +7 -7
  73. package/canon/tools/serena.md +6 -6
  74. package/dist/agents/process.js +3 -0
  75. package/dist/canon/manifest.js +2 -0
  76. package/dist/canon/skill-package.js +113 -0
  77. package/dist/canon/validate.js +16 -1
  78. package/dist/context/command.js +4 -1
  79. package/dist/context/context.js +6 -0
  80. package/dist/loop/dispatcher.js +1 -1
  81. package/dist/loop/loop.js +26 -0
  82. package/dist/loop/parallel-command.js +3 -0
  83. package/dist/loop/run-command.js +11 -0
  84. package/dist/loop/watchdog.js +28 -11
  85. package/dist/loop/worker.js +11 -0
  86. package/dist/prd/command.js +17 -17
  87. package/dist/retrofit/apply.js +22 -7
  88. package/dist/retrofit/command.js +4 -1
  89. package/dist/retrofit/config.js +4 -0
  90. package/dist/retrofit/context-actions.js +1 -1
  91. package/dist/retrofit/detect.js +2 -0
  92. package/dist/retrofit/planners/claude.js +16 -20
  93. package/dist/retrofit/planners/codex.js +3 -7
  94. package/dist/retrofit/planners/gemini.js +11 -1
  95. package/dist/retrofit/preserve.js +2 -2
  96. package/dist/retrofit/report.js +5 -0
  97. package/dist/retrofit/skill-actions.js +66 -0
  98. package/dist/retrofit/ui-detect.js +83 -0
  99. package/dist/scan/gate.js +36 -0
  100. package/docs/MIGRATING-TO-1.0.md +33 -33
  101. package/docs/MIGRATING-TO-1.1.md +27 -27
  102. package/docs/MIGRATING-TO-1.4.md +70 -70
  103. package/docs/PUBLISHING.md +91 -91
  104. package/docs/superpowers/plans/2026-06-28-baustein-e-context-layer.md +981 -981
  105. package/docs/superpowers/plans/2026-06-29-baustein-f-routing.md +258 -258
  106. package/docs/superpowers/plans/2026-06-29-baustein-g-loop-observability.md +1006 -1006
  107. package/docs/superpowers/plans/2026-06-29-baustein-h-loop-robustness.md +374 -374
  108. package/docs/superpowers/plans/2026-06-30-baustein-i-visual-design-verification.md +450 -450
  109. package/docs/superpowers/plans/2026-07-02-baustein-k-zero-to-100-bootstrap.md +1024 -1024
  110. package/docs/superpowers/plans/2026-07-02-baustein-m-flow-smoke-proofs.md +574 -574
  111. package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -537
  112. package/docs/superpowers/plans/2026-08-16-artifact-backed-output-compaction.md +329 -329
  113. package/docs/superpowers/plans/2026-08-20-automatic-ui-design-gate.md +59 -0
  114. package/docs/superpowers/plans/2026-08-20-capability-skills-and-context.md +51 -0
  115. package/docs/superpowers/plans/2026-08-20-complete-skill-packages-and-invocation.md +59 -0
  116. package/docs/superpowers/plans/2026-08-20-windows-reliability-and-release.md +67 -0
  117. package/docs/superpowers/specs/2026-06-28-baustein-e-context-layer-design.md +146 -146
  118. package/docs/superpowers/specs/2026-06-29-baustein-f-routing-design.md +106 -106
  119. package/docs/superpowers/specs/2026-06-29-baustein-g-loop-observability-design.md +186 -186
  120. package/docs/superpowers/specs/2026-06-29-baustein-h-loop-robustness-design.md +113 -113
  121. package/docs/superpowers/specs/2026-06-30-baustein-i-visual-design-verification-design.md +98 -98
  122. package/docs/superpowers/specs/2026-07-02-baustein-k-zero-to-100-bootstrap-design.md +200 -200
  123. package/docs/superpowers/specs/2026-07-02-baustein-m-flow-smoke-proofs-design.md +155 -155
  124. package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -422
  125. package/docs/superpowers/specs/2026-08-16-artifact-backed-output-compaction-design.md +166 -166
  126. package/docs/superpowers/specs/2026-08-20-skill-capabilities-and-reliability-design.md +391 -0
  127. package/gemini-extension.json +6 -6
  128. package/hooks/hooks.json +19 -19
  129. package/package.json +84 -84
package/TODOS.md CHANGED
@@ -1,5 +1,5 @@
1
- # Yoke follow-up work
2
-
3
- - Add provider-native output schemas when all three CLIs expose compatible stable APIs.
4
- - Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
5
- - Add signed provenance and attestations to npm and GitHub releases.
1
+ # Yoke follow-up work
2
+
3
+ - Add provider-native output schemas when all three CLIs expose compatible stable APIs.
4
+ - Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
5
+ - Add signed provenance and attestations to npm and GitHub releases.
package/agents/docs.toml CHANGED
@@ -1,6 +1,6 @@
1
- name = "docs"
2
- description = "Documentation specialist for release and API consistency."
3
- sandbox_mode = "workspace-write"
4
- developer_instructions = """
5
- Update only documentation required by the assigned change. Verify commands and version references against the repository.
6
- """
1
+ name = "docs"
2
+ description = "Documentation specialist for release and API consistency."
3
+ sandbox_mode = "workspace-write"
4
+ developer_instructions = """
5
+ Update only documentation required by the assigned change. Verify commands and version references against the repository.
6
+ """
@@ -1,6 +1,6 @@
1
- name = "implementer"
2
- description = "Implementation specialist for one scoped story."
3
- sandbox_mode = "workspace-write"
4
- developer_instructions = """
5
- Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
6
- """
1
+ name = "implementer"
2
+ description = "Implementation specialist for one scoped story."
3
+ sandbox_mode = "workspace-write"
4
+ developer_instructions = """
5
+ Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
6
+ """
@@ -1,6 +1,6 @@
1
- name = "reviewer"
2
- description = "Read-only reviewer for correctness and acceptance criteria."
3
- sandbox_mode = "read-only"
4
- developer_instructions = """
5
- Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.
6
- """
1
+ name = "reviewer"
2
+ description = "Read-only reviewer for correctness and acceptance criteria."
3
+ sandbox_mode = "read-only"
4
+ developer_instructions = """
5
+ Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.
6
+ """
@@ -1,6 +1,6 @@
1
- name = "security"
2
- description = "Read-only security reviewer for changed code."
3
- sandbox_mode = "read-only"
4
- developer_instructions = """
5
- Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.
6
- """
1
+ name = "security"
2
+ description = "Read-only security reviewer for changed code."
3
+ sandbox_mode = "read-only"
4
+ developer_instructions = """
5
+ Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.
6
+ """
package/bench/README.md CHANGED
@@ -1,86 +1,86 @@
1
- # Yoke benchmark — tokens · speed · quality
2
-
3
- Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
- the same PRD within each comparison, and three measured dimensions:
5
-
6
- Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
- multi-run performance study currently measures Codex only; provider support tests are not treated
8
- as performance evidence for Claude or Gemini.
9
-
10
- | Dimension | How it is measured |
11
- |---|---|
12
- | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
13
- | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
14
- | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
15
-
16
- ## The fixture (`fixtures/string-kit`)
17
-
18
- A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
19
- `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
20
- stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
21
- final quality check runs everything. No npm installs, so results measure the agent — not the
22
- network.
23
-
24
- ## The routing fixture (`fixtures/routing-queue`)
25
-
26
- A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
- `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
- letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
- enough that a lower-cost worker can potentially repay one routing-controller call per story.
30
-
31
- ## Running it
32
-
33
- ```bash
34
- npm run build
35
- node bench/run.mjs --runner=claude # or gemini / codex
36
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
- node bench/analyze-routing-study.mjs
43
- node bench/run-matrix.mjs --label=release-1.0
44
- node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
45
- ```
46
-
47
- Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
48
- explicit `--run-root`), git-inits it,
49
- drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
50
- `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
51
- The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
52
- failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
53
-
54
- ### Gate-output compaction benchmark
55
-
56
- `output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
57
- It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
58
- remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
59
- bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
60
- comparison to Aphrodite's corpus or published ratios.
61
-
62
- `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
63
- routing registry, junctions this checkout's dependencies, and replays the seed's original
64
- acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
65
- visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
66
- to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
67
- aggregates the checked-in three-pair Codex-only study.
68
-
69
- ## Caveats (read before quoting numbers)
70
-
71
- - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
72
- alternating-order pairs per arm. Re-run before setting broad policy defaults.
73
- - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
74
- policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
75
- to exclude personal MCP/plugin startup from both sides.
76
- - Model identity matters more than CLI identity: `tokens.model` records what actually served
77
- the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
78
- - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
79
- treat cached input as equivalent to newly processed input. Dollar cost is reported only when
80
- the provider emits it—Yoke does not guess prices from a model name.
81
- - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
82
- hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
83
- - Cumulative verify means a story's duration includes fixing any regressions it caused.
84
-
85
- Results live in [`results/`](results/) — one JSON per run, summarized in
86
- [`RESULTS.md`](RESULTS.md).
1
+ # Yoke benchmark — tokens · speed · quality
2
+
3
+ Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
+ the same PRD within each comparison, and three measured dimensions:
5
+
6
+ Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
+ multi-run performance study currently measures Codex only; provider support tests are not treated
8
+ as performance evidence for Claude or Gemini.
9
+
10
+ | Dimension | How it is measured |
11
+ |---|---|
12
+ | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
13
+ | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
14
+ | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
15
+
16
+ ## The fixture (`fixtures/string-kit`)
17
+
18
+ A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
19
+ `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
20
+ stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
21
+ final quality check runs everything. No npm installs, so results measure the agent — not the
22
+ network.
23
+
24
+ ## The routing fixture (`fixtures/routing-queue`)
25
+
26
+ A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
+ `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
+ letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
+ enough that a lower-cost worker can potentially repay one routing-controller call per story.
30
+
31
+ ## Running it
32
+
33
+ ```bash
34
+ npm run build
35
+ node bench/run.mjs --runner=claude # or gemini / codex
36
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
+ node bench/analyze-routing-study.mjs
43
+ node bench/run-matrix.mjs --label=release-1.0
44
+ node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
45
+ ```
46
+
47
+ Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
48
+ explicit `--run-root`), git-inits it,
49
+ drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
50
+ `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
51
+ The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
52
+ failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
53
+
54
+ ### Gate-output compaction benchmark
55
+
56
+ `output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
57
+ It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
58
+ remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
59
+ bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
60
+ comparison to Aphrodite's corpus or published ratios.
61
+
62
+ `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
63
+ routing registry, junctions this checkout's dependencies, and replays the seed's original
64
+ acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
65
+ visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
66
+ to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
67
+ aggregates the checked-in three-pair Codex-only study.
68
+
69
+ ## Caveats (read before quoting numbers)
70
+
71
+ - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
72
+ alternating-order pairs per arm. Re-run before setting broad policy defaults.
73
+ - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
74
+ policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
75
+ to exclude personal MCP/plugin startup from both sides.
76
+ - Model identity matters more than CLI identity: `tokens.model` records what actually served
77
+ the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
78
+ - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
79
+ treat cached input as equivalent to newly processed input. Dollar cost is reported only when
80
+ the provider emits it—Yoke does not guess prices from a model name.
81
+ - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
82
+ hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
83
+ - Cumulative verify means a story's duration includes fixing any regressions it caused.
84
+
85
+ Results live in [`results/`](results/) — one JSON per run, summarized in
86
+ [`RESULTS.md`](RESULTS.md).
package/bench/RESULTS.md CHANGED
@@ -1,22 +1,22 @@
1
- # Benchmark results
2
-
3
- Result schema v1 records fixture version, sample label, permission profile, telemetry/model
4
- availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
5
-
6
- Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
7
- [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
8
-
9
- ## Runs
10
-
11
- | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
12
- |---|---|---|---|---|---|---|---|---|
13
- | 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
14
- | 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
15
- | 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
16
- | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
17
- | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
18
- | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
19
-
1
+ # Benchmark results
2
+
3
+ Result schema v1 records fixture version, sample label, permission profile, telemetry/model
4
+ availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
5
+
6
+ Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
7
+ [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
8
+
9
+ ## Runs
10
+
11
+ | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
12
+ |---|---|---|---|---|---|---|---|---|
13
+ | 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
14
+ | 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
15
+ | 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
16
+ | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
17
+ | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
18
+ | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
19
+
20
20
  Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
21
21
  passed verify on the first iteration; the final quality check (all 16 assertions on the final
22
22
  tree, outside the loop) is green.
@@ -115,22 +115,22 @@ zero-token deterministic SELF fast path for clearly high-risk architecture/priva
115
115
  This is again N=1. Raw valid rows:
116
116
  [`routing off`](results/yoke-large-codex-routing-off-2026-08-01T22-58-01.json) and
117
117
  [`routing on`](results/yoke-large-codex-routing-on-2026-08-01T23-09-39.json).
118
-
119
- The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
120
- before changing the fixture, and the Codex executable could not be probed in this Windows
121
- environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
122
-
123
- ## What the harness caught before producing a single number
124
-
125
- Building an honest benchmark is itself a verification pass. The first runs found two real
126
- Yoke bugs, both fixed in 0.3.0:
127
-
128
- 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
129
- 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
130
- 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
131
- `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
132
- stdin (which selects headless mode) and passes only `--yolo`.
133
-
118
+
119
+ The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
120
+ before changing the fixture, and the Codex executable could not be probed in this Windows
121
+ environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
122
+
123
+ ## What the harness caught before producing a single number
124
+
125
+ Building an honest benchmark is itself a verification pass. The first runs found two real
126
+ Yoke bugs, both fixed in 0.3.0:
127
+
128
+ 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
129
+ 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
130
+ 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
131
+ `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
132
+ stdin (which selects headless mode) and passes only `--yolo`.
133
+
134
134
  ## Reading the numbers
135
135
 
136
136
  - Tokens come from provider telemetry when available; missing telemetry is recorded as missing,
@@ -1,65 +1,65 @@
1
- import { createHash } from 'node:crypto'
2
- import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
3
- import { tmpdir } from 'node:os'
4
- import { join, resolve } from 'node:path'
5
- import { fileURLToPath } from 'node:url'
6
-
7
- async function runtimeModules() {
8
- const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
9
- const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
10
- const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
11
- return Promise.all([
12
- import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
13
- import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
14
- ])
15
- }
16
-
17
- function fixture() {
18
- return [
19
- '=== stdout ===',
20
- 'compiling application',
21
- 'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
22
- 'the failing call is part of the checkout flow',
23
- ...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
24
- '=== stderr ===',
25
- 'Tests: 1 failed, 249 passed, 250 total',
26
- ].join('\n')
27
- }
28
-
29
- export async function runOutputCompactionBenchmark() {
30
- const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
31
- const raw = fixture()
32
- const previewBudgetBytes = 512
33
- const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
34
- const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
35
- try {
36
- const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
37
- const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
38
- const storedDigest = createHash('sha256').update(stored).digest('hex')
39
- const previewBytes = Buffer.byteLength(compacted.preview)
40
- const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
41
- return {
42
- fixture: 'gate-output-v1',
43
- rawBytes: Buffer.byteLength(raw),
44
- rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
45
- previewBudgetBytes,
46
- previewBytes,
47
- previewApproxTokens: Math.ceil(previewBytes / 4),
48
- referencedBytes,
49
- compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
50
- earlyErrorRetained: compacted.preview.includes('error TS2304'),
51
- finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
52
- digestRoundTrip: storedDigest === artifact.sha256,
53
- }
54
- } finally {
55
- rmSync(dir, { recursive: true, force: true })
56
- }
57
- }
58
-
59
- if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
60
- const result = await runOutputCompactionBenchmark()
61
- console.log(JSON.stringify(result, null, 2))
62
- if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
63
- process.exitCode = 1
64
- }
65
- }
1
+ import { createHash } from 'node:crypto'
2
+ import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
3
+ import { tmpdir } from 'node:os'
4
+ import { join, resolve } from 'node:path'
5
+ import { fileURLToPath } from 'node:url'
6
+
7
+ async function runtimeModules() {
8
+ const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
9
+ const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
10
+ const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
11
+ return Promise.all([
12
+ import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
13
+ import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
14
+ ])
15
+ }
16
+
17
+ function fixture() {
18
+ return [
19
+ '=== stdout ===',
20
+ 'compiling application',
21
+ 'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
22
+ 'the failing call is part of the checkout flow',
23
+ ...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
24
+ '=== stderr ===',
25
+ 'Tests: 1 failed, 249 passed, 250 total',
26
+ ].join('\n')
27
+ }
28
+
29
+ export async function runOutputCompactionBenchmark() {
30
+ const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
31
+ const raw = fixture()
32
+ const previewBudgetBytes = 512
33
+ const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
34
+ const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
35
+ try {
36
+ const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
37
+ const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
38
+ const storedDigest = createHash('sha256').update(stored).digest('hex')
39
+ const previewBytes = Buffer.byteLength(compacted.preview)
40
+ const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
41
+ return {
42
+ fixture: 'gate-output-v1',
43
+ rawBytes: Buffer.byteLength(raw),
44
+ rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
45
+ previewBudgetBytes,
46
+ previewBytes,
47
+ previewApproxTokens: Math.ceil(previewBytes / 4),
48
+ referencedBytes,
49
+ compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
50
+ earlyErrorRetained: compacted.preview.includes('error TS2304'),
51
+ finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
52
+ digestRoundTrip: storedDigest === artifact.sha256,
53
+ }
54
+ } finally {
55
+ rmSync(dir, { recursive: true, force: true })
56
+ }
57
+ }
58
+
59
+ if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
60
+ const result = await runOutputCompactionBenchmark()
61
+ console.log(JSON.stringify(result, null, 2))
62
+ if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
63
+ process.exitCode = 1
64
+ }
65
+ }
@@ -1,12 +1,12 @@
1
- export const requiredResultFields = [
2
- 'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
3
- 'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
4
- 'iterations', 'finalTestsPass',
5
- ]
6
-
7
- export function validateResult(result) {
8
- const missing = requiredResultFields.filter(key => !(key in result))
9
- if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
10
- if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
- return result
12
- }
1
+ export const requiredResultFields = [
2
+ 'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
3
+ 'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
4
+ 'iterations', 'finalTestsPass',
5
+ ]
6
+
7
+ export function validateResult(result) {
8
+ const missing = requiredResultFields.filter(key => !(key in result))
9
+ if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
10
+ if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
+ return result
12
+ }
@@ -1,50 +1,50 @@
1
- {
2
- "schemaVersion": 1,
3
- "fixtureVersion": "string-kit@1",
4
- "runner": "claude",
5
- "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
- "permissionProfile": "safe",
7
- "yokeVersion": "1.0.0",
8
- "fixture": "string-kit",
9
- "startedAt": "2026-07-27T18:03:26.598Z",
10
- "wallClockMs": 11345,
11
- "exitCode": 1,
12
- "finalState": "blocked",
13
- "verdict": "blocked",
14
- "blocker": "Claude runner exited before implementation; fixture verification remained red.",
15
- "conflicts": 0,
16
- "iterations": 1,
17
- "finalTestsPass": false,
18
- "progress": {
19
- "passed": 0,
20
- "total": 3
21
- },
22
- "usageAvailable": false,
23
- "modelAvailable": false,
24
- "tokens": {
25
- "inputTokens": 0,
26
- "outputTokens": 0,
27
- "model": "<synthetic>"
28
- },
29
- "stories": [
30
- {
31
- "id": "STORY-1",
32
- "durationMs": 10954,
33
- "iterations": 1,
34
- "finalTestsPass": false
35
- },
36
- {
37
- "id": "STORY-2",
38
- "durationMs": null,
39
- "iterations": 0,
40
- "finalTestsPass": false
41
- },
42
- {
43
- "id": "STORY-3",
44
- "durationMs": null,
45
- "iterations": 0,
46
- "finalTestsPass": false
47
- }
48
- ],
49
- "srcLoc": 3
50
- }
1
+ {
2
+ "schemaVersion": 1,
3
+ "fixtureVersion": "string-kit@1",
4
+ "runner": "claude",
5
+ "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
+ "permissionProfile": "safe",
7
+ "yokeVersion": "1.0.0",
8
+ "fixture": "string-kit",
9
+ "startedAt": "2026-07-27T18:03:26.598Z",
10
+ "wallClockMs": 11345,
11
+ "exitCode": 1,
12
+ "finalState": "blocked",
13
+ "verdict": "blocked",
14
+ "blocker": "Claude runner exited before implementation; fixture verification remained red.",
15
+ "conflicts": 0,
16
+ "iterations": 1,
17
+ "finalTestsPass": false,
18
+ "progress": {
19
+ "passed": 0,
20
+ "total": 3
21
+ },
22
+ "usageAvailable": false,
23
+ "modelAvailable": false,
24
+ "tokens": {
25
+ "inputTokens": 0,
26
+ "outputTokens": 0,
27
+ "model": "<synthetic>"
28
+ },
29
+ "stories": [
30
+ {
31
+ "id": "STORY-1",
32
+ "durationMs": 10954,
33
+ "iterations": 1,
34
+ "finalTestsPass": false
35
+ },
36
+ {
37
+ "id": "STORY-2",
38
+ "durationMs": null,
39
+ "iterations": 0,
40
+ "finalTestsPass": false
41
+ },
42
+ {
43
+ "id": "STORY-3",
44
+ "durationMs": null,
45
+ "iterations": 0,
46
+ "finalTestsPass": false
47
+ }
48
+ ],
49
+ "srcLoc": 3
50
+ }
@@ -1,15 +1,15 @@
1
- {
2
- "schemaVersion": 1,
3
- "fixtureVersion": "string-kit@1",
4
- "runner": "codex",
5
- "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
- "permissionProfile": "safe",
7
- "usageAvailable": false,
8
- "modelAvailable": false,
9
- "verdict": "unavailable",
10
- "blocker": "Zugriff verweigert",
11
- "conflicts": 0,
12
- "wallClockMs": null,
13
- "iterations": 0,
14
- "finalTestsPass": false
15
- }
1
+ {
2
+ "schemaVersion": 1,
3
+ "fixtureVersion": "string-kit@1",
4
+ "runner": "codex",
5
+ "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
+ "permissionProfile": "safe",
7
+ "usageAvailable": false,
8
+ "modelAvailable": false,
9
+ "verdict": "unavailable",
10
+ "blocker": "Zugriff verweigert",
11
+ "conflicts": 0,
12
+ "wallClockMs": null,
13
+ "iterations": 0,
14
+ "finalTestsPass": false
15
+ }