@hecer/yoke 0.9.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/.claude-plugin/marketplace.json +18 -0
  2. package/.claude-plugin/plugin.json +13 -0
  3. package/.codex-plugin/plugin.json +7 -0
  4. package/CHANGELOG.md +192 -149
  5. package/README.md +101 -51
  6. package/TODOS.md +8 -0
  7. package/agents/docs.toml +6 -0
  8. package/agents/implementer.toml +6 -0
  9. package/agents/reviewer.toml +6 -0
  10. package/agents/security.toml +6 -0
  11. package/bench/README.md +45 -42
  12. package/bench/RESULTS.md +46 -36
  13. package/bench/result-schema.mjs +12 -0
  14. package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
  15. package/bench/results/codex-unavailable-1785175418318.json +15 -0
  16. package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
  17. package/bench/run-matrix.mjs +26 -0
  18. package/bench/run.mjs +127 -115
  19. package/canon/AGENTS.md +2 -0
  20. package/canon/loop/loop-spec.md +4 -2
  21. package/canon/loop/prd.schema.md +5 -0
  22. package/canon/manifest.yaml +2 -1
  23. package/canon/skills/authoring-prd/SKILL.md +10 -3
  24. package/canon/skills/ship/SKILL.md +2 -7
  25. package/canon/skills/workflow/SKILL.md +4 -0
  26. package/canon/skills/yoke-retrofit/SKILL.md +18 -11
  27. package/canon/skills/yoke-workflow/SKILL.md +20 -0
  28. package/canon/tools/codex-rtk-hook.mjs +36 -0
  29. package/dist/agents/host.js +26 -0
  30. package/dist/agents/providers.js +23 -0
  31. package/dist/agents/telemetry.js +30 -0
  32. package/dist/agents/types.js +1 -0
  33. package/dist/audit/changes.js +6 -0
  34. package/dist/audit/command.js +64 -0
  35. package/dist/audit/dependencies.js +21 -0
  36. package/dist/audit/secrets.js +16 -0
  37. package/dist/audit/types.js +1 -0
  38. package/dist/cli.js +189 -6
  39. package/dist/context/context.js +15 -2
  40. package/dist/loop/claims.js +57 -0
  41. package/dist/loop/cleanup.js +98 -27
  42. package/dist/loop/decision.js +517 -0
  43. package/dist/loop/git.js +31 -2
  44. package/dist/loop/identity.js +27 -0
  45. package/dist/loop/lock.js +104 -13
  46. package/dist/loop/loop.js +49 -2
  47. package/dist/loop/merge-queue.js +20 -0
  48. package/dist/loop/parallel.js +39 -0
  49. package/dist/loop/prd.js +48 -2
  50. package/dist/loop/run-command.js +118 -12
  51. package/dist/loop/runner.js +48 -30
  52. package/dist/loop/scheduler.js +8 -0
  53. package/dist/prd/command.js +30 -21
  54. package/dist/retrofit/command.js +3 -2
  55. package/dist/retrofit/config.js +16 -0
  56. package/dist/retrofit/gitignore.js +8 -0
  57. package/dist/retrofit/planners/codex.js +64 -19
  58. package/dist/retrofit/report.js +1 -1
  59. package/dist/review/command.js +52 -12
  60. package/dist/review/verdict.js +45 -0
  61. package/dist/setup/command.js +82 -0
  62. package/docs/MIGRATING-TO-1.0.md +33 -0
  63. package/docs/MIGRATING-TO-1.1.md +27 -0
  64. package/docs/PUBLISHING.md +77 -41
  65. package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
  66. package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
  67. package/gemini-extension.json +6 -0
  68. package/hooks/hooks.json +19 -0
  69. package/package.json +84 -67
  70. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
  71. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
  72. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
  73. package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
  74. package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
  75. package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
  76. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
  77. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
  78. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
  79. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
  80. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
  81. package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
  82. package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
  83. package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
  84. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
  85. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
  86. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
package/README.md CHANGED
@@ -2,6 +2,11 @@
2
2
 
3
3
  # 🐂 Yoke
4
4
 
5
+ <!-- yoke:version:start -->1.1.0<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->559<!-- yoke:tests:end -->
7
+ <!-- yoke:skills:start -->29<!-- yoke:skills:end -->
8
+ <!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
9
+
5
10
  ### One harness, three agents — and zero trust in "done."
6
11
 
7
12
  **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Then, when you want it, an opt-in autonomous loop ships your spec story-by-story: tested, cross-model-reviewed, committed — **with a screenshot to prove every story and a video for every failure**.
@@ -12,7 +17,7 @@
12
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
13
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
14
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
15
- ![Tests](https://img.shields.io/badge/tests-439%20passing-brightgreen.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-559%20passing-brightgreen.svg)
16
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini-8A2BE2)
17
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
18
23
 
@@ -20,7 +25,13 @@
20
25
 
21
26
  </div>
22
27
 
23
- > **TL;DR** — `yoke new my-app --idea="..."` scaffolds a git repo, installs the harness for all three agents, and drafts a story backlog from your idea. `yoke loop run my-app --isolate --review` then implements it story by story behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. If any gate is red, nothing is committed. When a story is done, there's a photo of it in `.yoke/proof/<story>/`.
28
+ > **TL;DR** — `yoke setup .` asks five questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it story by story behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. If any gate is red, nothing is committed. When a story is done, there's a photo of it in `.yoke/proof/<story>/`.
29
+
30
+ Yoke 1.1 is safe-by-default: provider CLIs use autonomous sandbox profiles unless `--unsafe`
31
+ is explicit; reviews require a schema-valid verdict and a different model unless
32
+ `--allow-self-review` is explicit; commits enforce the human identity from project config or Git.
33
+ See [the 1.1 migration guide](docs/MIGRATING-TO-1.1.md) for setup/decision parity and
34
+ [the 1.0 guide](docs/MIGRATING-TO-1.0.md) for the earlier safety-policy changes.
24
35
 
25
36
  ---
26
37
 
@@ -33,7 +44,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
33
44
  | 🎭 **The verification gap** — *"agent says done, but it isn't"* | Agents submit confidently on 100% of runs while resolving far fewer; "all tests pass" when they were never run ([silent-failures research](https://arxiv.org/pdf/2603.25764)) | The loop trusts **your verify command's exit code**, never the agent's word. A story is `passes: true` only after tests are green, the reviewer approved, and the commit landed — atomically. Plus: **screenshot proofs** per story. |
34
45
  | 🔀 **Three agents, three configs** | Teams hand-maintain `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, skills, and MCP wiring separately — copy-paste drift everywhere | **One canon → `yoke retrofit`** generates the idiomatic native artifacts for each agent. Change the canon once, re-retrofit everywhere. |
35
46
  | 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
36
- | 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a *second* model reviews the diff as a pass/fail exit-code gate — chainable into verify, pre-push, or CI. Cross-model review measurably catches what self-review misses. |
47
+ | 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
37
48
 
38
49
  **Who it's for:** anyone driving Claude Code, Codex CLI, or Gemini CLI on real projects — especially if you use more than one, want autonomous runs you can trust, or are tired of "done" meaning "probably". Greenfield (`yoke new`) and brownfield (`yoke retrofit`) both work.
39
50
 
@@ -61,7 +72,7 @@ $ ls reading-app/.yoke/proof/STORY-2/
61
72
  home.png list.png # photographic evidence, labelled per story
62
73
  ```
63
74
 
64
- Every claim in that transcript is enforced by code paths with tests behind them — 439 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
75
+ Every claim in that transcript is enforced by code paths with tests behind them — 559 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
65
76
 
66
77
  ## 🚀 Quickstart
67
78
 
@@ -74,15 +85,14 @@ yoke new my-app --idea="a CLI that tracks reading lists"
74
85
  yoke loop on my-app && yoke loop run my-app --isolate
75
86
 
76
87
  # — or retrofit an existing project —
77
- yoke validate canon # 1) sanity-check the canon
78
- yoke retrofit /path/to/project --agent=all # 2) install (non-destructive)
79
- yoke loop on /path/to/project # 3) optional: the autonomous loop
88
+ yoke setup /path/to/project # interactive: agents, graph, loop, runner, decisions
89
+ yoke validate canon # sanity-check the canon
80
90
  yoke loop run /path/to/project --isolate --reviewer=codex --max=20
81
91
  ```
82
92
 
83
93
  > Requires Node ≥ 20 and git. No global install? `node /path/to/yoke/dist/cli.js …` or `npm --prefix /path/to/yoke run yoke -- …` work too. The MCP tools (rtk, graphify/Serena, Playwright MCP) are wired by Yoke but installed separately — the generated config is a clearly-labelled, adjustable template.
84
94
 
85
- ### Or install the skills as a Claude Code plugin
95
+ ### Skills before the first setup
86
96
 
87
97
  The canon is also packaged as a Claude Code plugin — the repo is its own marketplace:
88
98
 
@@ -93,6 +103,12 @@ The canon is also packaged as a Claude Code plugin — the repo is its own marke
93
103
 
94
104
  That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
95
105
 
106
+ For Codex, no preinstalled skill is required: run `npx @hecer/yoke setup .` in a terminal, or
107
+ ask Codex to run the five-question Yoke setup flow. The retrofit writes native skills to
108
+ `.agents/skills/`, including `yoke-retrofit` and `yoke-workflow`; start a fresh Codex task if an
109
+ already-open task does not discover newly installed skills. The npm package also contains
110
+ `.codex-plugin/plugin.json` for Codex plugin hosts.
111
+
96
112
  ### Staying up to date
97
113
 
98
114
  Yoke checks for new releases npm/gh-style: a **non-blocking background check** (at most once a day, detached, offline-safe) prints a one-line hint when a newer version exists — upgrading itself is always an explicit act:
@@ -114,11 +130,11 @@ Auto-upgrade is deliberately **not** the default: a gate harness shouldn't chang
114
130
 
115
131
  Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, and Gemini CLI):
116
132
 
117
- > **Set it up** — *"Install the Yoke harness in this project: run `yoke retrofit . --agent=all`, pick the code-graph you'd recommend for this codebase, and leave the autonomous loop disabled for now. Then summarise what changed and commit it."*
133
+ > **Set it up** — *"Set up Yoke in this project. Ask me the Yoke setup questions one at a time with your recommendation, then run `yoke setup . --yes` with the selected host, agents, code graph, loop, runner, and decision policy. Commit in my configured identity."*
118
134
 
119
135
  > **Work the disciplined way** — *"From now on follow the Yoke skills you just installed: brainstorm → spec → plan → TDD → review before merging. Use the `review` skill before any merge."*
120
136
 
121
- > **Run autonomously** — *"Write `.yoke/prd.yaml` with one story per task (each needs acceptance criteria — see the `authoring-prd` skill), set `verify.command` in `.yoke/config.yaml`, enable the loop with `yoke loop on .`, then run it in small visible batches: `yoke loop run . --max=5`. After each batch show me `yoke loop status .`."*
137
+ > **Plan, then run autonomously** — *"Use the `yoke-workflow` skill. Ask only the planning questions that materially change the product, write the approved plan and loop-ready stories, then execute every approved story without routine follow-ups. Follow the configured `auto` or `critical` decision policy."*
122
138
 
123
139
  > **Watch / unblock** — *"Run `yoke loop status .`. If it says BLOCKED, run the project's verify command, find the root cause, fix it without weakening tests, then continue the loop."*
124
140
 
@@ -132,14 +148,16 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
132
148
 
133
149
  | Command | What it does | Exit codes |
134
150
  |---|---|---|
151
+ | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop]` | Shared five-question setup for Claude, Codex, and Gemini; `--yes` applies supplied/default choices non-interactively | `0` · `1` invalid setup |
135
152
  | `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
136
153
  | `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
137
154
  | `yoke retrofit [dir] [--agent=claude,codex,gemini\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness, non-destructively | `0` |
138
155
  | `yoke prd draft [dir] --idea= [--runner=] [--force]` | Idea → 5–12 stories with testable acceptance criteria | `0` · `1` invalid/guarded · `2` agent unavailable |
139
- | `yoke prd check [dir]` | PRD lint gate (schema, duplicate ids, empty acceptance) | `0` valid · `1` violations |
156
+ | `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
140
157
  | `yoke context init\|status [dir]` | Durable context layer (`PROJECT/DECISIONS/KNOWLEDGE.md`) | `0` |
141
- | `yoke loop on\|off\|status\|run\|cleanup [dir]` | The autonomous loop (see below) | run: `0` complete · `1` blocked/cap · `2` not runnable / already locked · `3` paused |
142
- | `yoke review [dir] [--reviewer=] [--base=] [--focus=]` | A **second model** reviews your diff | `0` approved · `1` findings · `2` no reviewer CLI |
158
+ | `yoke loop on\|off\|status\|decision\|answer\|resume\|run\|cleanup [dir]` | Autonomous loop; `decision` shows a critical stop, `answer` records it and resumes, `resume` retries a failed restart with the preserved safety options; cleanup deletes worktrees only with `--remove-worktrees` | run: `0` complete · `1` blocked/cap · `2` not runnable / already locked · `3` paused |
159
+ | `yoke review [dir] [--reviewer=] [--base=] [--focus=] [--json] [--allow-self-review]` | An independent model writes a schema-valid verdict | `0` approved · `1` findings/invalid verdict · `2` no independent reviewer |
160
+ | `yoke audit [dir] [--json]` | Dependency, high-confidence secret, and sensitive-change audit | `0` green · `1` blocking findings · `2` not runnable |
143
161
  | `yoke design-scan [dir] [--max=N] [--report]` | Static AI-slop design gate | `0` within budget · `1` over |
144
162
  | `yoke flow-smoke [dir] [--url=] [--label=]` | Browser gate with screenshot/video proofs | `0` green · `1` failures · `2` not runnable |
145
163
 
@@ -156,7 +174,7 @@ Three excellent projects, three different jobs. Honest version:
156
174
  | **Enforcement** | Advisory — skills *describe* the discipline; following them is up to the agent | Skill-driven; browser QA is genuinely real | **Mechanical** — gates live in code: clean tree, acceptance criteria, green tests, review verdict, commit integrity |
157
175
  | **Autonomy** | Interactive sessions | Interactive slash-commands (`/qa`, `/ship`, …) | Opt-in **Ralph loop** with watchdog, worktree isolation, single-flight lock, per-story proofs |
158
176
  | **Visual QA** | — | **Best-in-class**: live browser daemon (Chromium/CDP) with deep interactive QA | Built-in `flow-smoke` gate: screenshots always, video on failure, labelled per story — lighter, but *enforced* and cross-agent |
159
- | **Cross-model review** | — | `/codex` second opinion (Codex-only direction) | `yoke review` — resolves **codex → gemini → claude**, exit-code gate, works in and outside the loop |
177
+ | **Cross-model review** | — | `/codex` second opinion (Codex-only direction) | `yoke review` — resolves an independent provider and validates a structured verdict, inside or outside the loop |
160
178
  | **Footprint** | Markdown skills (plugin) | ~230 MB with browser runtime; hourly auto-update | Node CLI + markdown canon; Playwright only if you use flow-smoke, resolved **from your project** |
161
179
  | **License** | MIT | MIT | MIT |
162
180
 
@@ -189,10 +207,10 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
189
207
  | Agent | Artifacts |
190
208
  |---|---|
191
209
  | **Claude** | `.claude/skills/`, `AGENTS.md`, `CLAUDE.md`, `.mcp.json` (code-graph + Playwright), and an rtk `PreToolUse` hook when WSL is available |
192
- | **Codex** | `AGENTS.md` (native), `.codex/config.toml` (MCP servers), `RTK.md` |
210
+ | **Codex** | `.agents/skills/`, `AGENTS.md`, `RTK.md`, `.codex/config.toml`, native hooks, reusable `.codex/agents/*.toml`, and package plugin metadata |
193
211
  | **Gemini** | `GEMINI.md`, `.gemini/commands/*.toml` (one per skill, full body), `.gemini/settings.json` (MCP + `AGENTS.md` context) |
194
212
 
195
- > **rtk asymmetry, handled:** Claude can rewrite commands transparently via a hook (needs WSL on Windows); Codex and Gemini have no such hook, so they get an instruction to prefix commands with `rtk` instead.
213
+ > **rtk integration:** Claude receives its PreToolUse hook; Codex receives a native hook adapter around `rtk hook check`; Gemini retains instruction-mode fallback where its CLI has no equivalent command-rewrite lifecycle.
196
214
 
197
215
  > **Composes with gstack:** if [gstack](https://github.com/garrytan/gstack) is installed (repo-local or global), `yoke retrofit` adds a short "Composed tools" routing note to **CLAUDE.md only** — telling Claude to prefer gstack's skills for capabilities Yoke doesn't ship (live-browser QA `/qa`, security audit `/cso`, ship/deploy `/ship`). No bundling, no dependency; the note is never written to the Codex or Gemini artifacts.
198
216
 
@@ -203,7 +221,7 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
203
221
  > instructions (tech stack, workflow, `@`-includes) inside it. Works in any yoke-written file;
204
222
  > content *outside* the markers is still replaced (and backed up under `.yoke/backup/`).
205
223
 
206
- ## 🧰 What's in the canon — 28 skills
224
+ ## 🧰 What's in the canon — 29 skills
207
225
 
208
226
  `yoke retrofit` installs all of these into each agent natively. Provenance is credited in [`canon/skills/ATTRIBUTION.md`](canon/skills/ATTRIBUTION.md).
209
227
 
@@ -239,11 +257,12 @@ To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.
239
257
  | `retro` | Engineering retrospective from commit history |
240
258
  | `document-release` | Post-ship documentation sync (README / CHANGELOG / …) |
241
259
 
242
- **Yoke-native** — *authored or adapted for this harness (8)*
260
+ **Yoke-native** — *authored or adapted for this harness (9)*
243
261
 
244
262
  | Skill | What it does |
245
263
  |---|---|
246
264
  | `yoke-retrofit` | Set up the Yoke harness in a project (detect → plan → apply) |
265
+ | `yoke-workflow` | Provider-neutral planning questions → approved PRD → autonomous stories → critical-decision resume |
247
266
  | `authoring-prd` | Slice a product idea into loop-ready stories with testable acceptance criteria |
248
267
  | `minimal-code` | Write the least code that solves the task (YAGNI; ponytail-derived) |
249
268
  | `performance` | Efficiency as a measured requirement: benchmarks as tests, budgets as gates, optimizations local + documented |
@@ -267,7 +286,7 @@ existing projects), then: creates and `git init`s the directory, writes a minima
267
286
  **context layer** (with `--idea` seeded into `PROJECT.md` as the north star), writes a commented
268
287
  **PRD template** to `.yoke/prd.yaml`, and makes the initial commit — so `--isolate` works from
269
288
  iteration 1. With `--idea`, it then drafts the PRD from your idea via an agent (`--runner=`,
270
- default `claude`) and commits it as a second commit (`docs: draft PRD from idea`).
289
+ the configured runner or active host) and commits it as a second commit (`docs: draft PRD from idea`).
271
290
 
272
291
  - **Exit codes** — `0` success; `1` usage / non-empty dir / draft failure (the scaffold survives —
273
292
  retry with `yoke prd draft`); `2` requested draft agent unavailable.
@@ -276,16 +295,18 @@ default `claude`) and commits it as a second commit (`docs: draft PRD from idea`
276
295
  stories with testable behavioral acceptance criteria (greenfield STORY-1 scaffolds the project
277
296
  skeleton + test suite and wires `verify.command`). An existing PRD with stories is never
278
297
  overwritten without `--force`; the untouched template doesn't trigger the guard. Runs through
279
- the same idle-timeout watchdog as the loop (`--timeout`).
298
+ the same idle-timeout watchdog as the loop (`--timeout`). If `.yoke/plan.md` exists, its approved
299
+ goals, non-goals, constraints, and decisions are injected as settled context instead of being
300
+ reopened by the drafting agent.
280
301
 
281
302
  **`yoke prd check [dir]`** is the chainable pre-loop lint gate: schema validation plus
282
- duplicate-id, empty-acceptance, and zero-stories checks. Exits `0` with
303
+ duplicate-id, empty-acceptance, unresolved-placeholder, and zero-stories checks. Exits `0` with
283
304
  `✓ PRD valid — N stories, M pass`, `1` on any violation. The `authoring-prd` canon skill
284
305
  teaches interactive sessions the same story-slicing discipline.
285
306
 
286
307
  ## 🤖 The autonomous loop
287
308
 
288
- Opt-in and off by default. Each iteration starts a **fresh agent** and passes through hard gates before anything is committed:
309
+ Opt-in; `yoke setup` recommends enabling it for new installs, while `retrofit` alone keeps it off unless requested. Each iteration starts a **fresh agent** and passes through hard gates before anything is committed:
289
310
 
290
311
  ```mermaid
291
312
  flowchart LR
@@ -309,7 +330,7 @@ yoke loop run . \
309
330
  --runner=codex \ # implement with Codex…
310
331
  --reviewer=claude \ # …review with Claude (role separation)
311
332
  --isolate \ # each story in a throwaway git worktree
312
- --on-ambiguity=abort \ # strict: stop on undecidable criteria instead of guessing
333
+ --decision-policy=critical \ # pause only for high-impact decisions; routine choices stay autonomous
313
334
  --max=20
314
335
  yoke loop off . # disable
315
336
  ```
@@ -369,25 +390,51 @@ A per-iteration **idle timeout** guards against a genuinely hung agent: if the a
369
390
  output is **never** killed — the output stream *is* the liveness signal. Set a project default
370
391
  with `loop.timeoutMinutes` in `.yoke/config.yaml`.
371
392
 
372
- ### Ambiguous stories: questions belong in planning
393
+ ### Decision policy: autonomous by default, interrupt only when configured
373
394
 
374
- A loop run has nobody to ask, so the runner prompt always forbids questions. What the agent
375
- does when an acceptance criterion is genuinely ambiguous is configurable:
395
+ Planning questions happen before the loop. The provider-neutral `yoke-workflow` skill asks only
396
+ questions whose answer materially changes product behavior, scope, architecture, security, data
397
+ ownership, external cost, or an irreversible choice. It saves the approved brief in
398
+ `.yoke/plan.md`; `yoke prd draft` consumes it, and `yoke prd check` rejects explicit unresolved
399
+ placeholders such as `TBD`.
376
400
 
377
- - **Default (`resolve`) — never stop:** the agent picks the interpretation most consistent
378
- with the other criteria and the existing code, states it in its final message, and the loop
379
- keeps going.
380
- - **Strict (`abort`):** the agent must not guess — it writes its open question(s) to
381
- `.yoke/ambiguity.md` and stops. The loop consumes that file, skips verify (an unimplemented
382
- story would otherwise sail through on pre-existing green tests), and blocks with the
383
- question in the reason, e.g.
384
- `story S6 stopped: ambiguous acceptance criteria — Which auth provider should S6 use?`
385
- Answer by sharpening the story's acceptance criteria, then re-run.
401
+ The unattended loop then follows `loop.decisionPolicy`:
402
+
403
+ ```yaml
404
+ loop:
405
+ enabled: true
406
+ decisionPolicy: critical # or auto
407
+ runner:
408
+ agent: codex # setup chooses the current host by default
409
+ ```
410
+
411
+ - **`auto` (default):** routine ambiguity and implementation details are resolved using the
412
+ approved plan, acceptance criteria, current code, and project conventions. The loop does not
413
+ ask follow-up questions.
414
+ - **`critical`:** routine choices are still resolved automatically. Only high-impact decisions
415
+ involving public architecture, security/privacy, destructive migration or data loss, material
416
+ external cost, legal/compliance exposure, or another irreversible choice may pause the story.
417
+ The agent writes a schema-validated request; the loop blocks before verify and preserves it as
418
+ `.yoke/pending-decision.yaml`.
419
+
420
+ Inspect and answer a critical stop:
421
+
422
+ ```bash
423
+ yoke loop decision .
424
+ yoke loop answer . --choice=A --rationale="Matches the existing identity model"
425
+ ```
386
426
 
387
- Enable strict mode per run with `yoke loop run . --on-ambiguity=abort` or per project with
388
- `loop.onAmbiguity: abort` in `.yoke/config.yaml`. Either way, the cheapest fix is upstream:
389
- put every clarifying question into the PRD **before** the loop starts (`yoke prd draft`
390
- criteria must be testable and decision-free).
427
+ `answer` validates the choice against the still-open story, appends it to
428
+ `.yoke/context/DECISIONS.md`, commits only that file using the configured human identity, clears
429
+ the pending request, and resumes the same story with the original runner, isolation, review,
430
+ permission, timeout, JSON, decision-policy, and iteration settings intact. Add
431
+ `--no-resume` when a supervisor should restart the loop separately. If the automatic restart
432
+ cannot begin because a provider/reviewer is unavailable or another process owns the lock, run
433
+ `yoke loop resume .`; its request-bound options are retained under Git's private state directory
434
+ until a loop actually runs. To intentionally abandon an orphaned or stale private resume state,
435
+ use `yoke loop resume . --discard`; pending decisions are never deleted by that command. Existing
436
+ `loop.onAmbiguity: resolve|abort` and `--on-ambiguity=` remain supported as compatibility aliases;
437
+ new projects should use `decisionPolicy: auto|critical`.
391
438
 
392
439
  ### Performance budgets: efficiency as a gate, not a style
393
440
 
@@ -418,21 +465,26 @@ committed even if the agent process exited non-zero (a common Windows `.cmd`-wra
418
465
  A failing verify is retried up to `verify.retries` times (default 1) so a transient flake
419
466
  self-heals while a real failure still blocks.
420
467
 
421
- `.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, `.yoke/story-durations.json`,
422
- and `.yoke/ambiguity.md` are runtime artifacts; `yoke retrofit` gitignores them (along with
468
+ `.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, its takeover/recovery leases, lock/decision temp files, `.yoke/story-durations.json`,
469
+ `.yoke/ambiguity.md`, and the critical-decision request/answering files are runtime artifacts;
470
+ `yoke retrofit` gitignores them (along with
423
471
  `.yoke/worktrees/`, `.yoke/backup/`, and `.yoke/proof/`) so they never trip the clean-tree gate.
424
472
 
425
473
  ### Single-flight guard + cleanup
426
474
 
427
475
  Two concurrent `yoke loop run`s would race on the PRD and status files, so the loop takes a
428
- **lock** (`.yoke/loop.lock`) for the duration of a run. A second invocation exits `2` with
476
+ **lock** (`.yoke/loop.lock`) for the duration of a run. Complete lock metadata is published atomically;
477
+ stale takeover is serialized by `.yoke/loop.lock.takeover`. A second invocation exits `2` with
429
478
  `Another loop is already running here (pid …). If that is wrong, run: yoke loop cleanup`. A lock
430
479
  whose holder process is dead is taken over automatically (with a warning).
431
480
 
432
481
  **`yoke loop cleanup [dir]`** removes what a crashed loop leaves behind: every worktree under
433
482
  `.yoke/worktrees/` (via `git worktree remove --force` + `prune` — user-created worktrees are
434
483
  never touched) and a **stale** lock file. A live lock is reported and left alone. Exits `0`
435
- when everything cleaned, `1` if any removal failed.
484
+ when everything cleaned, `1` if any removal failed. If a machine/process crash leaves the cleanup
485
+ recovery lease itself behind, an operator can run
486
+ `yoke loop cleanup . --discard-stale-recovery`; Yoke refuses while its recorded PID is alive, and
487
+ the force flag must not be run concurrently.
436
488
 
437
489
  ## 🔍 Cross-model review (`yoke review`)
438
490
 
@@ -516,7 +568,8 @@ Yoke keeps durable, cross-session context so a fresh-context agent is never blin
516
568
 
517
569
  `yoke retrofit` scaffolds these files (non-destructively — your edits are never overwritten).
518
570
  The loop reads them into every agent + reviewer prompt and logs decisions back on each story's
519
- commit. Manage them directly with `yoke context init` and `yoke context status`. The
571
+ commit. Decision history is explicitly delimited as untrusted reference data, so stored text is
572
+ never treated as fresh instructions. Manage the files directly with `yoke context init` and `yoke context status`. The
520
573
  `maintaining-context` skill teaches agents to honour the same files during interactive work.
521
574
 
522
575
  > Commit `.yoke/context/` to git. The `--isolate` loop runs each iteration in a worktree
@@ -587,17 +640,14 @@ docs/superpowers/ # the spec and every component's implementation plan
587
640
 
588
641
  ## 🗺️ Roadmap
589
642
 
590
- - **npm publish** (`@hecer/yoke`) — one-liner `npx` install (package prepared).
591
- - **Security gate** — a `cso`-style audit skill + `yoke audit` (deps, secrets, diff surface).
592
- - **Token-budget gate** — per-story budget with abort; cost transparency for loop runs.
593
- - **Multi-reviewer quorum** — N independent reviewers with distinct lenses (correctness / security / acceptance).
594
- - **Merge queue** — re-test against the latest main before integrating, for parallel/multi-agent loops.
595
- - **More agents** — the canon→retrofit pattern generalises; OpenCode and Copilot CLI are natural next targets.
643
+ Yoke 1.1's completed release work moved to the changelog. Remaining, explicitly scoped work
644
+ is tracked in [`TODOS.md`](TODOS.md), including provider subprocess wiring for the tested
645
+ parallel dispatcher, broader benchmark samples, native output schemas, and release provenance.
596
646
 
597
647
  ## 🧪 Development
598
648
 
599
649
  ```bash
600
- npm test # vitest (322 tests)
650
+ npm test # vitest (559 tests)
601
651
  npm run build # tsc, no emit errors
602
652
  npm run yoke -- validate canon
603
653
  ```
package/TODOS.md ADDED
@@ -0,0 +1,8 @@
1
+ # Yoke follow-up work
2
+
3
+ - Wire the tested async parallel dispatcher to provider subprocess workers. Until then the CLI
4
+ rejects `--parallel=N` for `N > 1`; scheduler, claims, and merge queue APIs are available
5
+ without claiming a CLI speed-up.
6
+ - Add provider-native output schemas when all three CLIs expose compatible stable APIs.
7
+ - Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
8
+ - Add signed provenance and attestations to npm and GitHub releases.
@@ -0,0 +1,6 @@
1
+ name = "docs"
2
+ description = "Documentation specialist for release and API consistency."
3
+ sandbox_mode = "workspace-write"
4
+ developer_instructions = """
5
+ Update only documentation required by the assigned change. Verify commands and version references against the repository.
6
+ """
@@ -0,0 +1,6 @@
1
+ name = "implementer"
2
+ description = "Implementation specialist for one scoped story."
3
+ sandbox_mode = "workspace-write"
4
+ developer_instructions = """
5
+ Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
6
+ """
@@ -0,0 +1,6 @@
1
+ name = "reviewer"
2
+ description = "Read-only reviewer for correctness and acceptance criteria."
3
+ sandbox_mode = "read-only"
4
+ developer_instructions = """
5
+ Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.
6
+ """
@@ -0,0 +1,6 @@
1
+ name = "security"
2
+ description = "Read-only security reviewer for changed code."
3
+ sandbox_mode = "read-only"
4
+ developer_instructions = """
5
+ Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.
6
+ """
package/bench/README.md CHANGED
@@ -1,42 +1,45 @@
1
- # Yoke benchmark — tokens · speed · quality
2
-
3
- Reproducible cross-runner benchmark for the Yoke loop. One fixed fixture project, the same
4
- PRD for every runner, three measured dimensions:
5
-
6
- | Dimension | How it is measured |
7
- |---|---|
8
- | **Tokens** | The loop's own token hook (`.yoke/loop-status.json`, claude runner via `--output-format stream-json`; model id included). Gemini/Codex runners do not report usage yet — recorded as `null`, a documented gap. |
9
- | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
10
- | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
11
-
12
- ## The fixture (`fixtures/string-kit`)
13
-
14
- A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
15
- `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
16
- stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
17
- final quality check runs everything. No npm installs, so results measure the agent — not the
18
- network.
19
-
20
- ## Running it
21
-
22
- ```bash
23
- npm run build
24
- node bench/run.mjs --runner=claude # or gemini / codex
25
- ```
26
-
27
- Each run copies the fixture to `bench/.runs/<runner>-<stamp>` (git-ignored), git-inits it,
28
- drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
29
- `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
30
-
31
- ## Caveats (read before quoting numbers)
32
-
33
- - **N=1 per run.** Agent runs are stochastic; treat single runs as indicative, not
34
- statistically robust. Re-run and compare.
35
- - Model identity matters more than CLI identity: `tokens.model` records what actually served
36
- the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
37
- - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
38
- hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
39
- - Cumulative verify means a story's duration includes fixing any regressions it caused.
40
-
41
- Results live in [`results/`](results/) — one JSON per run, summarized in
42
- [`RESULTS.md`](RESULTS.md).
1
+ # Yoke benchmark — tokens · speed · quality
2
+
3
+ Reproducible cross-runner benchmark for the Yoke loop. One fixed fixture project, the same
4
+ PRD for every runner, three measured dimensions:
5
+
6
+ | Dimension | How it is measured |
7
+ |---|---|
8
+ | **Tokens** | The loop's own token hook (`.yoke/loop-status.json`, claude runner via `--output-format stream-json`; model id included). Gemini/Codex runners do not report usage yet — recorded as `null`, a documented gap. |
9
+ | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
10
+ | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
11
+
12
+ ## The fixture (`fixtures/string-kit`)
13
+
14
+ A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
15
+ `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
16
+ stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
17
+ final quality check runs everything. No npm installs, so results measure the agent — not the
18
+ network.
19
+
20
+ ## Running it
21
+
22
+ ```bash
23
+ npm run build
24
+ node bench/run.mjs --runner=claude # or gemini / codex
25
+ node bench/run-matrix.mjs --label=release-1.0
26
+ ```
27
+
28
+ Each run copies the fixture to `bench/.runs/<runner>-<stamp>` (git-ignored), git-inits it,
29
+ drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
30
+ `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
31
+ The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
32
+ failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
33
+
34
+ ## Caveats (read before quoting numbers)
35
+
36
+ - **N=1 per run.** Agent runs are stochastic; treat single runs as indicative, not
37
+ statistically robust. Re-run and compare.
38
+ - Model identity matters more than CLI identity: `tokens.model` records what actually served
39
+ the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
40
+ - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
41
+ hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
42
+ - Cumulative verify means a story's duration includes fixing any regressions it caused.
43
+
44
+ Results live in [`results/`](results/) — one JSON per run, summarized in
45
+ [`RESULTS.md`](RESULTS.md).
package/bench/RESULTS.md CHANGED
@@ -1,36 +1,46 @@
1
- # Benchmark results
2
-
3
- Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
4
- [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
5
-
6
- ## Runs
7
-
8
- | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
9
- |---|---|---|---|---|---|---|---|---|
10
- | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
11
- | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
12
- | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
13
-
14
- Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
15
- passed verify on the first iteration; the final quality check (all 16 assertions on the final
16
- tree, outside the loop) is green.
17
-
18
- ## What the harness caught before producing a single number
19
-
20
- Building an honest benchmark is itself a verification pass. The first runs found two real
21
- Yoke bugs, both fixed in 0.3.0:
22
-
23
- 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
24
- 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
25
- 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
26
- `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
27
- stdin (which selects headless mode) and passes only `--yolo`.
28
-
29
- ## Reading the numbers
30
-
31
- - Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
32
- open gap, see the multi-agent design doc).
33
- - N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
34
- compare rows.
35
- - ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
36
- gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
1
+ # Benchmark results
2
+
3
+ Result schema v1 records fixture version, sample label, permission profile, telemetry/model
4
+ availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
5
+
6
+ Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
7
+ [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
8
+
9
+ ## Runs
10
+
11
+ | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
12
+ |---|---|---|---|---|---|---|---|---|
13
+ | 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
14
+ | 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
15
+ | 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
16
+ | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
17
+ | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
18
+ | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
19
+
20
+ Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
21
+ passed verify on the first iteration; the final quality check (all 16 assertions on the final
22
+ tree, outside the loop) is green.
23
+
24
+ The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
25
+ before changing the fixture, and the Codex executable could not be probed in this Windows
26
+ environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
27
+
28
+ ## What the harness caught before producing a single number
29
+
30
+ Building an honest benchmark is itself a verification pass. The first runs found two real
31
+ Yoke bugs, both fixed in 0.3.0:
32
+
33
+ 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
34
+ 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
35
+ 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
36
+ `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
37
+ stdin (which selects headless mode) and passes only `--yolo`.
38
+
39
+ ## Reading the numbers
40
+
41
+ - Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
42
+ open gap, see the multi-agent design doc).
43
+ - N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
44
+ compare rows.
45
+ - ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
46
+ gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
@@ -0,0 +1,12 @@
1
+ export const requiredResultFields = [
2
+ 'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
3
+ 'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
4
+ 'iterations', 'finalTestsPass',
5
+ ]
6
+
7
+ export function validateResult(result) {
8
+ const missing = requiredResultFields.filter(key => !(key in result))
9
+ if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
10
+ if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
+ return result
12
+ }