@hecer/yoke 0.9.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +18 -0
- package/.claude-plugin/plugin.json +13 -0
- package/.codex-plugin/plugin.json +7 -0
- package/CHANGELOG.md +192 -149
- package/README.md +101 -51
- package/TODOS.md +8 -0
- package/agents/docs.toml +6 -0
- package/agents/implementer.toml +6 -0
- package/agents/reviewer.toml +6 -0
- package/agents/security.toml +6 -0
- package/bench/README.md +45 -42
- package/bench/RESULTS.md +46 -36
- package/bench/result-schema.mjs +12 -0
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
- package/bench/results/codex-unavailable-1785175418318.json +15 -0
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
- package/bench/run-matrix.mjs +26 -0
- package/bench/run.mjs +127 -115
- package/canon/AGENTS.md +2 -0
- package/canon/loop/loop-spec.md +4 -2
- package/canon/loop/prd.schema.md +5 -0
- package/canon/manifest.yaml +2 -1
- package/canon/skills/authoring-prd/SKILL.md +10 -3
- package/canon/skills/ship/SKILL.md +2 -7
- package/canon/skills/workflow/SKILL.md +4 -0
- package/canon/skills/yoke-retrofit/SKILL.md +18 -11
- package/canon/skills/yoke-workflow/SKILL.md +20 -0
- package/canon/tools/codex-rtk-hook.mjs +36 -0
- package/dist/agents/host.js +26 -0
- package/dist/agents/providers.js +23 -0
- package/dist/agents/telemetry.js +30 -0
- package/dist/agents/types.js +1 -0
- package/dist/audit/changes.js +6 -0
- package/dist/audit/command.js +64 -0
- package/dist/audit/dependencies.js +21 -0
- package/dist/audit/secrets.js +16 -0
- package/dist/audit/types.js +1 -0
- package/dist/cli.js +189 -6
- package/dist/context/context.js +15 -2
- package/dist/loop/claims.js +57 -0
- package/dist/loop/cleanup.js +98 -27
- package/dist/loop/decision.js +517 -0
- package/dist/loop/git.js +31 -2
- package/dist/loop/identity.js +27 -0
- package/dist/loop/lock.js +104 -13
- package/dist/loop/loop.js +49 -2
- package/dist/loop/merge-queue.js +20 -0
- package/dist/loop/parallel.js +39 -0
- package/dist/loop/prd.js +48 -2
- package/dist/loop/run-command.js +118 -12
- package/dist/loop/runner.js +48 -30
- package/dist/loop/scheduler.js +8 -0
- package/dist/prd/command.js +30 -21
- package/dist/retrofit/command.js +3 -2
- package/dist/retrofit/config.js +16 -0
- package/dist/retrofit/gitignore.js +8 -0
- package/dist/retrofit/planners/codex.js +64 -19
- package/dist/retrofit/report.js +1 -1
- package/dist/review/command.js +52 -12
- package/dist/review/verdict.js +45 -0
- package/dist/setup/command.js +82 -0
- package/docs/MIGRATING-TO-1.0.md +33 -0
- package/docs/MIGRATING-TO-1.1.md +27 -0
- package/docs/PUBLISHING.md +77 -41
- package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
- package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
- package/gemini-extension.json +6 -0
- package/hooks/hooks.json +19 -0
- package/package.json +84 -67
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
- package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
- package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
- package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
- package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
- package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
package/README.md
CHANGED
|
@@ -2,6 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
# 🐂 Yoke
|
|
4
4
|
|
|
5
|
+
<!-- yoke:version:start -->1.1.0<!-- yoke:version:end -->
|
|
6
|
+
<!-- yoke:tests:start -->559<!-- yoke:tests:end -->
|
|
7
|
+
<!-- yoke:skills:start -->29<!-- yoke:skills:end -->
|
|
8
|
+
<!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
|
|
9
|
+
|
|
5
10
|
### One harness, three agents — and zero trust in "done."
|
|
6
11
|
|
|
7
12
|
**Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Then, when you want it, an opt-in autonomous loop ships your spec story-by-story: tested, cross-model-reviewed, committed — **with a screenshot to prove every story and a video for every failure**.
|
|
@@ -12,7 +17,7 @@
|
|
|
12
17
|
[](#-license)
|
|
13
18
|

|
|
14
19
|

|
|
15
|
-

|
|
16
21
|

|
|
17
22
|

|
|
18
23
|
|
|
@@ -20,7 +25,13 @@
|
|
|
20
25
|
|
|
21
26
|
</div>
|
|
22
27
|
|
|
23
|
-
> **TL;DR** — `yoke new my-app --idea="..."`
|
|
28
|
+
> **TL;DR** — `yoke setup .` asks five questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it story by story behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. If any gate is red, nothing is committed. When a story is done, there's a photo of it in `.yoke/proof/<story>/`.
|
|
29
|
+
|
|
30
|
+
Yoke 1.1 is safe-by-default: provider CLIs use autonomous sandbox profiles unless `--unsafe`
|
|
31
|
+
is explicit; reviews require a schema-valid verdict and a different model unless
|
|
32
|
+
`--allow-self-review` is explicit; commits enforce the human identity from project config or Git.
|
|
33
|
+
See [the 1.1 migration guide](docs/MIGRATING-TO-1.1.md) for setup/decision parity and
|
|
34
|
+
[the 1.0 guide](docs/MIGRATING-TO-1.0.md) for the earlier safety-policy changes.
|
|
24
35
|
|
|
25
36
|
---
|
|
26
37
|
|
|
@@ -33,7 +44,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
|
|
|
33
44
|
| 🎭 **The verification gap** — *"agent says done, but it isn't"* | Agents submit confidently on 100% of runs while resolving far fewer; "all tests pass" when they were never run ([silent-failures research](https://arxiv.org/pdf/2603.25764)) | The loop trusts **your verify command's exit code**, never the agent's word. A story is `passes: true` only after tests are green, the reviewer approved, and the commit landed — atomically. Plus: **screenshot proofs** per story. |
|
|
34
45
|
| 🔀 **Three agents, three configs** | Teams hand-maintain `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, skills, and MCP wiring separately — copy-paste drift everywhere | **One canon → `yoke retrofit`** generates the idiomatic native artifacts for each agent. Change the canon once, re-retrofit everywhere. |
|
|
35
46
|
| 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
|
|
36
|
-
| 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a
|
|
47
|
+
| 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
|
|
37
48
|
|
|
38
49
|
**Who it's for:** anyone driving Claude Code, Codex CLI, or Gemini CLI on real projects — especially if you use more than one, want autonomous runs you can trust, or are tired of "done" meaning "probably". Greenfield (`yoke new`) and brownfield (`yoke retrofit`) both work.
|
|
39
50
|
|
|
@@ -61,7 +72,7 @@ $ ls reading-app/.yoke/proof/STORY-2/
|
|
|
61
72
|
home.png list.png # photographic evidence, labelled per story
|
|
62
73
|
```
|
|
63
74
|
|
|
64
|
-
Every claim in that transcript is enforced by code paths with tests behind them —
|
|
75
|
+
Every claim in that transcript is enforced by code paths with tests behind them — 559 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
|
|
65
76
|
|
|
66
77
|
## 🚀 Quickstart
|
|
67
78
|
|
|
@@ -74,15 +85,14 @@ yoke new my-app --idea="a CLI that tracks reading lists"
|
|
|
74
85
|
yoke loop on my-app && yoke loop run my-app --isolate
|
|
75
86
|
|
|
76
87
|
# — or retrofit an existing project —
|
|
77
|
-
yoke
|
|
78
|
-
yoke
|
|
79
|
-
yoke loop on /path/to/project # 3) optional: the autonomous loop
|
|
88
|
+
yoke setup /path/to/project # interactive: agents, graph, loop, runner, decisions
|
|
89
|
+
yoke validate canon # sanity-check the canon
|
|
80
90
|
yoke loop run /path/to/project --isolate --reviewer=codex --max=20
|
|
81
91
|
```
|
|
82
92
|
|
|
83
93
|
> Requires Node ≥ 20 and git. No global install? `node /path/to/yoke/dist/cli.js …` or `npm --prefix /path/to/yoke run yoke -- …` work too. The MCP tools (rtk, graphify/Serena, Playwright MCP) are wired by Yoke but installed separately — the generated config is a clearly-labelled, adjustable template.
|
|
84
94
|
|
|
85
|
-
###
|
|
95
|
+
### Skills before the first setup
|
|
86
96
|
|
|
87
97
|
The canon is also packaged as a Claude Code plugin — the repo is its own marketplace:
|
|
88
98
|
|
|
@@ -93,6 +103,12 @@ The canon is also packaged as a Claude Code plugin — the repo is its own marke
|
|
|
93
103
|
|
|
94
104
|
That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
|
|
95
105
|
|
|
106
|
+
For Codex, no preinstalled skill is required: run `npx @hecer/yoke setup .` in a terminal, or
|
|
107
|
+
ask Codex to run the five-question Yoke setup flow. The retrofit writes native skills to
|
|
108
|
+
`.agents/skills/`, including `yoke-retrofit` and `yoke-workflow`; start a fresh Codex task if an
|
|
109
|
+
already-open task does not discover newly installed skills. The npm package also contains
|
|
110
|
+
`.codex-plugin/plugin.json` for Codex plugin hosts.
|
|
111
|
+
|
|
96
112
|
### Staying up to date
|
|
97
113
|
|
|
98
114
|
Yoke checks for new releases npm/gh-style: a **non-blocking background check** (at most once a day, detached, offline-safe) prints a one-line hint when a newer version exists — upgrading itself is always an explicit act:
|
|
@@ -114,11 +130,11 @@ Auto-upgrade is deliberately **not** the default: a gate harness shouldn't chang
|
|
|
114
130
|
|
|
115
131
|
Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, and Gemini CLI):
|
|
116
132
|
|
|
117
|
-
> **Set it up** — *"
|
|
133
|
+
> **Set it up** — *"Set up Yoke in this project. Ask me the Yoke setup questions one at a time with your recommendation, then run `yoke setup . --yes` with the selected host, agents, code graph, loop, runner, and decision policy. Commit in my configured identity."*
|
|
118
134
|
|
|
119
135
|
> **Work the disciplined way** — *"From now on follow the Yoke skills you just installed: brainstorm → spec → plan → TDD → review before merging. Use the `review` skill before any merge."*
|
|
120
136
|
|
|
121
|
-
> **
|
|
137
|
+
> **Plan, then run autonomously** — *"Use the `yoke-workflow` skill. Ask only the planning questions that materially change the product, write the approved plan and loop-ready stories, then execute every approved story without routine follow-ups. Follow the configured `auto` or `critical` decision policy."*
|
|
122
138
|
|
|
123
139
|
> **Watch / unblock** — *"Run `yoke loop status .`. If it says BLOCKED, run the project's verify command, find the root cause, fix it without weakening tests, then continue the loop."*
|
|
124
140
|
|
|
@@ -132,14 +148,16 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
|
|
|
132
148
|
|
|
133
149
|
| Command | What it does | Exit codes |
|
|
134
150
|
|---|---|---|
|
|
151
|
+
| `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop]` | Shared five-question setup for Claude, Codex, and Gemini; `--yes` applies supplied/default choices non-interactively | `0` · `1` invalid setup |
|
|
135
152
|
| `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
|
|
136
153
|
| `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
|
|
137
154
|
| `yoke retrofit [dir] [--agent=claude,codex,gemini\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness, non-destructively | `0` |
|
|
138
155
|
| `yoke prd draft [dir] --idea= [--runner=] [--force]` | Idea → 5–12 stories with testable acceptance criteria | `0` · `1` invalid/guarded · `2` agent unavailable |
|
|
139
|
-
| `yoke prd check [dir]` | PRD lint gate (schema, duplicate ids,
|
|
156
|
+
| `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
|
|
140
157
|
| `yoke context init\|status [dir]` | Durable context layer (`PROJECT/DECISIONS/KNOWLEDGE.md`) | `0` |
|
|
141
|
-
| `yoke loop on\|off\|status\|run\|cleanup [dir]` |
|
|
142
|
-
| `yoke review [dir] [--reviewer=] [--base=] [--focus=]` |
|
|
158
|
+
| `yoke loop on\|off\|status\|decision\|answer\|resume\|run\|cleanup [dir]` | Autonomous loop; `decision` shows a critical stop, `answer` records it and resumes, `resume` retries a failed restart with the preserved safety options; cleanup deletes worktrees only with `--remove-worktrees` | run: `0` complete · `1` blocked/cap · `2` not runnable / already locked · `3` paused |
|
|
159
|
+
| `yoke review [dir] [--reviewer=] [--base=] [--focus=] [--json] [--allow-self-review]` | An independent model writes a schema-valid verdict | `0` approved · `1` findings/invalid verdict · `2` no independent reviewer |
|
|
160
|
+
| `yoke audit [dir] [--json]` | Dependency, high-confidence secret, and sensitive-change audit | `0` green · `1` blocking findings · `2` not runnable |
|
|
143
161
|
| `yoke design-scan [dir] [--max=N] [--report]` | Static AI-slop design gate | `0` within budget · `1` over |
|
|
144
162
|
| `yoke flow-smoke [dir] [--url=] [--label=]` | Browser gate with screenshot/video proofs | `0` green · `1` failures · `2` not runnable |
|
|
145
163
|
|
|
@@ -156,7 +174,7 @@ Three excellent projects, three different jobs. Honest version:
|
|
|
156
174
|
| **Enforcement** | Advisory — skills *describe* the discipline; following them is up to the agent | Skill-driven; browser QA is genuinely real | **Mechanical** — gates live in code: clean tree, acceptance criteria, green tests, review verdict, commit integrity |
|
|
157
175
|
| **Autonomy** | Interactive sessions | Interactive slash-commands (`/qa`, `/ship`, …) | Opt-in **Ralph loop** with watchdog, worktree isolation, single-flight lock, per-story proofs |
|
|
158
176
|
| **Visual QA** | — | **Best-in-class**: live browser daemon (Chromium/CDP) with deep interactive QA | Built-in `flow-smoke` gate: screenshots always, video on failure, labelled per story — lighter, but *enforced* and cross-agent |
|
|
159
|
-
| **Cross-model review** | — | `/codex` second opinion (Codex-only direction) | `yoke review` — resolves
|
|
177
|
+
| **Cross-model review** | — | `/codex` second opinion (Codex-only direction) | `yoke review` — resolves an independent provider and validates a structured verdict, inside or outside the loop |
|
|
160
178
|
| **Footprint** | Markdown skills (plugin) | ~230 MB with browser runtime; hourly auto-update | Node CLI + markdown canon; Playwright only if you use flow-smoke, resolved **from your project** |
|
|
161
179
|
| **License** | MIT | MIT | MIT |
|
|
162
180
|
|
|
@@ -189,10 +207,10 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
|
|
|
189
207
|
| Agent | Artifacts |
|
|
190
208
|
|---|---|
|
|
191
209
|
| **Claude** | `.claude/skills/`, `AGENTS.md`, `CLAUDE.md`, `.mcp.json` (code-graph + Playwright), and an rtk `PreToolUse` hook when WSL is available |
|
|
192
|
-
| **Codex** | `AGENTS.md`
|
|
210
|
+
| **Codex** | `.agents/skills/`, `AGENTS.md`, `RTK.md`, `.codex/config.toml`, native hooks, reusable `.codex/agents/*.toml`, and package plugin metadata |
|
|
193
211
|
| **Gemini** | `GEMINI.md`, `.gemini/commands/*.toml` (one per skill, full body), `.gemini/settings.json` (MCP + `AGENTS.md` context) |
|
|
194
212
|
|
|
195
|
-
> **rtk
|
|
213
|
+
> **rtk integration:** Claude receives its PreToolUse hook; Codex receives a native hook adapter around `rtk hook check`; Gemini retains instruction-mode fallback where its CLI has no equivalent command-rewrite lifecycle.
|
|
196
214
|
|
|
197
215
|
> **Composes with gstack:** if [gstack](https://github.com/garrytan/gstack) is installed (repo-local or global), `yoke retrofit` adds a short "Composed tools" routing note to **CLAUDE.md only** — telling Claude to prefer gstack's skills for capabilities Yoke doesn't ship (live-browser QA `/qa`, security audit `/cso`, ship/deploy `/ship`). No bundling, no dependency; the note is never written to the Codex or Gemini artifacts.
|
|
198
216
|
|
|
@@ -203,7 +221,7 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
|
|
|
203
221
|
> instructions (tech stack, workflow, `@`-includes) inside it. Works in any yoke-written file;
|
|
204
222
|
> content *outside* the markers is still replaced (and backed up under `.yoke/backup/`).
|
|
205
223
|
|
|
206
|
-
## 🧰 What's in the canon —
|
|
224
|
+
## 🧰 What's in the canon — 29 skills
|
|
207
225
|
|
|
208
226
|
`yoke retrofit` installs all of these into each agent natively. Provenance is credited in [`canon/skills/ATTRIBUTION.md`](canon/skills/ATTRIBUTION.md).
|
|
209
227
|
|
|
@@ -239,11 +257,12 @@ To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.
|
|
|
239
257
|
| `retro` | Engineering retrospective from commit history |
|
|
240
258
|
| `document-release` | Post-ship documentation sync (README / CHANGELOG / …) |
|
|
241
259
|
|
|
242
|
-
**Yoke-native** — *authored or adapted for this harness (
|
|
260
|
+
**Yoke-native** — *authored or adapted for this harness (9)*
|
|
243
261
|
|
|
244
262
|
| Skill | What it does |
|
|
245
263
|
|---|---|
|
|
246
264
|
| `yoke-retrofit` | Set up the Yoke harness in a project (detect → plan → apply) |
|
|
265
|
+
| `yoke-workflow` | Provider-neutral planning questions → approved PRD → autonomous stories → critical-decision resume |
|
|
247
266
|
| `authoring-prd` | Slice a product idea into loop-ready stories with testable acceptance criteria |
|
|
248
267
|
| `minimal-code` | Write the least code that solves the task (YAGNI; ponytail-derived) |
|
|
249
268
|
| `performance` | Efficiency as a measured requirement: benchmarks as tests, budgets as gates, optimizations local + documented |
|
|
@@ -267,7 +286,7 @@ existing projects), then: creates and `git init`s the directory, writes a minima
|
|
|
267
286
|
**context layer** (with `--idea` seeded into `PROJECT.md` as the north star), writes a commented
|
|
268
287
|
**PRD template** to `.yoke/prd.yaml`, and makes the initial commit — so `--isolate` works from
|
|
269
288
|
iteration 1. With `--idea`, it then drafts the PRD from your idea via an agent (`--runner=`,
|
|
270
|
-
|
|
289
|
+
the configured runner or active host) and commits it as a second commit (`docs: draft PRD from idea`).
|
|
271
290
|
|
|
272
291
|
- **Exit codes** — `0` success; `1` usage / non-empty dir / draft failure (the scaffold survives —
|
|
273
292
|
retry with `yoke prd draft`); `2` requested draft agent unavailable.
|
|
@@ -276,16 +295,18 @@ default `claude`) and commits it as a second commit (`docs: draft PRD from idea`
|
|
|
276
295
|
stories with testable behavioral acceptance criteria (greenfield STORY-1 scaffolds the project
|
|
277
296
|
skeleton + test suite and wires `verify.command`). An existing PRD with stories is never
|
|
278
297
|
overwritten without `--force`; the untouched template doesn't trigger the guard. Runs through
|
|
279
|
-
the same idle-timeout watchdog as the loop (`--timeout`).
|
|
298
|
+
the same idle-timeout watchdog as the loop (`--timeout`). If `.yoke/plan.md` exists, its approved
|
|
299
|
+
goals, non-goals, constraints, and decisions are injected as settled context instead of being
|
|
300
|
+
reopened by the drafting agent.
|
|
280
301
|
|
|
281
302
|
**`yoke prd check [dir]`** is the chainable pre-loop lint gate: schema validation plus
|
|
282
|
-
duplicate-id, empty-acceptance, and zero-stories checks. Exits `0` with
|
|
303
|
+
duplicate-id, empty-acceptance, unresolved-placeholder, and zero-stories checks. Exits `0` with
|
|
283
304
|
`✓ PRD valid — N stories, M pass`, `1` on any violation. The `authoring-prd` canon skill
|
|
284
305
|
teaches interactive sessions the same story-slicing discipline.
|
|
285
306
|
|
|
286
307
|
## 🤖 The autonomous loop
|
|
287
308
|
|
|
288
|
-
Opt-in
|
|
309
|
+
Opt-in; `yoke setup` recommends enabling it for new installs, while `retrofit` alone keeps it off unless requested. Each iteration starts a **fresh agent** and passes through hard gates before anything is committed:
|
|
289
310
|
|
|
290
311
|
```mermaid
|
|
291
312
|
flowchart LR
|
|
@@ -309,7 +330,7 @@ yoke loop run . \
|
|
|
309
330
|
--runner=codex \ # implement with Codex…
|
|
310
331
|
--reviewer=claude \ # …review with Claude (role separation)
|
|
311
332
|
--isolate \ # each story in a throwaway git worktree
|
|
312
|
-
--
|
|
333
|
+
--decision-policy=critical \ # pause only for high-impact decisions; routine choices stay autonomous
|
|
313
334
|
--max=20
|
|
314
335
|
yoke loop off . # disable
|
|
315
336
|
```
|
|
@@ -369,25 +390,51 @@ A per-iteration **idle timeout** guards against a genuinely hung agent: if the a
|
|
|
369
390
|
output is **never** killed — the output stream *is* the liveness signal. Set a project default
|
|
370
391
|
with `loop.timeoutMinutes` in `.yoke/config.yaml`.
|
|
371
392
|
|
|
372
|
-
###
|
|
393
|
+
### Decision policy: autonomous by default, interrupt only when configured
|
|
373
394
|
|
|
374
|
-
|
|
375
|
-
|
|
395
|
+
Planning questions happen before the loop. The provider-neutral `yoke-workflow` skill asks only
|
|
396
|
+
questions whose answer materially changes product behavior, scope, architecture, security, data
|
|
397
|
+
ownership, external cost, or an irreversible choice. It saves the approved brief in
|
|
398
|
+
`.yoke/plan.md`; `yoke prd draft` consumes it, and `yoke prd check` rejects explicit unresolved
|
|
399
|
+
placeholders such as `TBD`.
|
|
376
400
|
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
401
|
+
The unattended loop then follows `loop.decisionPolicy`:
|
|
402
|
+
|
|
403
|
+
```yaml
|
|
404
|
+
loop:
|
|
405
|
+
enabled: true
|
|
406
|
+
decisionPolicy: critical # or auto
|
|
407
|
+
runner:
|
|
408
|
+
agent: codex # setup chooses the current host by default
|
|
409
|
+
```
|
|
410
|
+
|
|
411
|
+
- **`auto` (default):** routine ambiguity and implementation details are resolved using the
|
|
412
|
+
approved plan, acceptance criteria, current code, and project conventions. The loop does not
|
|
413
|
+
ask follow-up questions.
|
|
414
|
+
- **`critical`:** routine choices are still resolved automatically. Only high-impact decisions
|
|
415
|
+
involving public architecture, security/privacy, destructive migration or data loss, material
|
|
416
|
+
external cost, legal/compliance exposure, or another irreversible choice may pause the story.
|
|
417
|
+
The agent writes a schema-validated request; the loop blocks before verify and preserves it as
|
|
418
|
+
`.yoke/pending-decision.yaml`.
|
|
419
|
+
|
|
420
|
+
Inspect and answer a critical stop:
|
|
421
|
+
|
|
422
|
+
```bash
|
|
423
|
+
yoke loop decision .
|
|
424
|
+
yoke loop answer . --choice=A --rationale="Matches the existing identity model"
|
|
425
|
+
```
|
|
386
426
|
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
427
|
+
`answer` validates the choice against the still-open story, appends it to
|
|
428
|
+
`.yoke/context/DECISIONS.md`, commits only that file using the configured human identity, clears
|
|
429
|
+
the pending request, and resumes the same story with the original runner, isolation, review,
|
|
430
|
+
permission, timeout, JSON, decision-policy, and iteration settings intact. Add
|
|
431
|
+
`--no-resume` when a supervisor should restart the loop separately. If the automatic restart
|
|
432
|
+
cannot begin because a provider/reviewer is unavailable or another process owns the lock, run
|
|
433
|
+
`yoke loop resume .`; its request-bound options are retained under Git's private state directory
|
|
434
|
+
until a loop actually runs. To intentionally abandon an orphaned or stale private resume state,
|
|
435
|
+
use `yoke loop resume . --discard`; pending decisions are never deleted by that command. Existing
|
|
436
|
+
`loop.onAmbiguity: resolve|abort` and `--on-ambiguity=` remain supported as compatibility aliases;
|
|
437
|
+
new projects should use `decisionPolicy: auto|critical`.
|
|
391
438
|
|
|
392
439
|
### Performance budgets: efficiency as a gate, not a style
|
|
393
440
|
|
|
@@ -418,21 +465,26 @@ committed even if the agent process exited non-zero (a common Windows `.cmd`-wra
|
|
|
418
465
|
A failing verify is retried up to `verify.retries` times (default 1) so a transient flake
|
|
419
466
|
self-heals while a real failure still blocks.
|
|
420
467
|
|
|
421
|
-
`.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, `.yoke/story-durations.json`,
|
|
422
|
-
|
|
468
|
+
`.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, its takeover/recovery leases, lock/decision temp files, `.yoke/story-durations.json`,
|
|
469
|
+
`.yoke/ambiguity.md`, and the critical-decision request/answering files are runtime artifacts;
|
|
470
|
+
`yoke retrofit` gitignores them (along with
|
|
423
471
|
`.yoke/worktrees/`, `.yoke/backup/`, and `.yoke/proof/`) so they never trip the clean-tree gate.
|
|
424
472
|
|
|
425
473
|
### Single-flight guard + cleanup
|
|
426
474
|
|
|
427
475
|
Two concurrent `yoke loop run`s would race on the PRD and status files, so the loop takes a
|
|
428
|
-
**lock** (`.yoke/loop.lock`) for the duration of a run.
|
|
476
|
+
**lock** (`.yoke/loop.lock`) for the duration of a run. Complete lock metadata is published atomically;
|
|
477
|
+
stale takeover is serialized by `.yoke/loop.lock.takeover`. A second invocation exits `2` with
|
|
429
478
|
`Another loop is already running here (pid …). If that is wrong, run: yoke loop cleanup`. A lock
|
|
430
479
|
whose holder process is dead is taken over automatically (with a warning).
|
|
431
480
|
|
|
432
481
|
**`yoke loop cleanup [dir]`** removes what a crashed loop leaves behind: every worktree under
|
|
433
482
|
`.yoke/worktrees/` (via `git worktree remove --force` + `prune` — user-created worktrees are
|
|
434
483
|
never touched) and a **stale** lock file. A live lock is reported and left alone. Exits `0`
|
|
435
|
-
when everything cleaned, `1` if any removal failed.
|
|
484
|
+
when everything cleaned, `1` if any removal failed. If a machine/process crash leaves the cleanup
|
|
485
|
+
recovery lease itself behind, an operator can run
|
|
486
|
+
`yoke loop cleanup . --discard-stale-recovery`; Yoke refuses while its recorded PID is alive, and
|
|
487
|
+
the force flag must not be run concurrently.
|
|
436
488
|
|
|
437
489
|
## 🔍 Cross-model review (`yoke review`)
|
|
438
490
|
|
|
@@ -516,7 +568,8 @@ Yoke keeps durable, cross-session context so a fresh-context agent is never blin
|
|
|
516
568
|
|
|
517
569
|
`yoke retrofit` scaffolds these files (non-destructively — your edits are never overwritten).
|
|
518
570
|
The loop reads them into every agent + reviewer prompt and logs decisions back on each story's
|
|
519
|
-
commit.
|
|
571
|
+
commit. Decision history is explicitly delimited as untrusted reference data, so stored text is
|
|
572
|
+
never treated as fresh instructions. Manage the files directly with `yoke context init` and `yoke context status`. The
|
|
520
573
|
`maintaining-context` skill teaches agents to honour the same files during interactive work.
|
|
521
574
|
|
|
522
575
|
> Commit `.yoke/context/` to git. The `--isolate` loop runs each iteration in a worktree
|
|
@@ -587,17 +640,14 @@ docs/superpowers/ # the spec and every component's implementation plan
|
|
|
587
640
|
|
|
588
641
|
## 🗺️ Roadmap
|
|
589
642
|
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
- **Multi-reviewer quorum** — N independent reviewers with distinct lenses (correctness / security / acceptance).
|
|
594
|
-
- **Merge queue** — re-test against the latest main before integrating, for parallel/multi-agent loops.
|
|
595
|
-
- **More agents** — the canon→retrofit pattern generalises; OpenCode and Copilot CLI are natural next targets.
|
|
643
|
+
Yoke 1.1's completed release work moved to the changelog. Remaining, explicitly scoped work
|
|
644
|
+
is tracked in [`TODOS.md`](TODOS.md), including provider subprocess wiring for the tested
|
|
645
|
+
parallel dispatcher, broader benchmark samples, native output schemas, and release provenance.
|
|
596
646
|
|
|
597
647
|
## 🧪 Development
|
|
598
648
|
|
|
599
649
|
```bash
|
|
600
|
-
npm test # vitest (
|
|
650
|
+
npm test # vitest (559 tests)
|
|
601
651
|
npm run build # tsc, no emit errors
|
|
602
652
|
npm run yoke -- validate canon
|
|
603
653
|
```
|
package/TODOS.md
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Yoke follow-up work
|
|
2
|
+
|
|
3
|
+
- Wire the tested async parallel dispatcher to provider subprocess workers. Until then the CLI
|
|
4
|
+
rejects `--parallel=N` for `N > 1`; scheduler, claims, and merge queue APIs are available
|
|
5
|
+
without claiming a CLI speed-up.
|
|
6
|
+
- Add provider-native output schemas when all three CLIs expose compatible stable APIs.
|
|
7
|
+
- Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
|
|
8
|
+
- Add signed provenance and attestations to npm and GitHub releases.
|
package/agents/docs.toml
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
name = "docs"
|
|
2
|
+
description = "Documentation specialist for release and API consistency."
|
|
3
|
+
sandbox_mode = "workspace-write"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Update only documentation required by the assigned change. Verify commands and version references against the repository.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
name = "implementer"
|
|
2
|
+
description = "Implementation specialist for one scoped story."
|
|
3
|
+
sandbox_mode = "workspace-write"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
|
|
6
|
+
"""
|
package/bench/README.md
CHANGED
|
@@ -1,42 +1,45 @@
|
|
|
1
|
-
# Yoke benchmark — tokens · speed · quality
|
|
2
|
-
|
|
3
|
-
Reproducible cross-runner benchmark for the Yoke loop. One fixed fixture project, the same
|
|
4
|
-
PRD for every runner, three measured dimensions:
|
|
5
|
-
|
|
6
|
-
| Dimension | How it is measured |
|
|
7
|
-
|---|---|
|
|
8
|
-
| **Tokens** | The loop's own token hook (`.yoke/loop-status.json`, claude runner via `--output-format stream-json`; model id included). Gemini/Codex runners do not report usage yet — recorded as `null`, a documented gap. |
|
|
9
|
-
| **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
|
|
10
|
-
| **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
|
|
11
|
-
|
|
12
|
-
## The fixture (`fixtures/string-kit`)
|
|
13
|
-
|
|
14
|
-
A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
|
|
15
|
-
`node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
|
|
16
|
-
stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
|
|
17
|
-
final quality check runs everything. No npm installs, so results measure the agent — not the
|
|
18
|
-
network.
|
|
19
|
-
|
|
20
|
-
## Running it
|
|
21
|
-
|
|
22
|
-
```bash
|
|
23
|
-
npm run build
|
|
24
|
-
node bench/run.mjs --runner=claude # or gemini / codex
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
`
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
1
|
+
# Yoke benchmark — tokens · speed · quality
|
|
2
|
+
|
|
3
|
+
Reproducible cross-runner benchmark for the Yoke loop. One fixed fixture project, the same
|
|
4
|
+
PRD for every runner, three measured dimensions:
|
|
5
|
+
|
|
6
|
+
| Dimension | How it is measured |
|
|
7
|
+
|---|---|
|
|
8
|
+
| **Tokens** | The loop's own token hook (`.yoke/loop-status.json`, claude runner via `--output-format stream-json`; model id included). Gemini/Codex runners do not report usage yet — recorded as `null`, a documented gap. |
|
|
9
|
+
| **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
|
|
10
|
+
| **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
|
|
11
|
+
|
|
12
|
+
## The fixture (`fixtures/string-kit`)
|
|
13
|
+
|
|
14
|
+
A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
|
|
15
|
+
`node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
|
|
16
|
+
stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
|
|
17
|
+
final quality check runs everything. No npm installs, so results measure the agent — not the
|
|
18
|
+
network.
|
|
19
|
+
|
|
20
|
+
## Running it
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
npm run build
|
|
24
|
+
node bench/run.mjs --runner=claude # or gemini / codex
|
|
25
|
+
node bench/run-matrix.mjs --label=release-1.0
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Each run copies the fixture to `bench/.runs/<runner>-<stamp>` (git-ignored), git-inits it,
|
|
29
|
+
drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
|
|
30
|
+
`bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
|
|
31
|
+
The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
|
|
32
|
+
failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
|
|
33
|
+
|
|
34
|
+
## Caveats (read before quoting numbers)
|
|
35
|
+
|
|
36
|
+
- **N=1 per run.** Agent runs are stochastic; treat single runs as indicative, not
|
|
37
|
+
statistically robust. Re-run and compare.
|
|
38
|
+
- Model identity matters more than CLI identity: `tokens.model` records what actually served
|
|
39
|
+
the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
|
|
40
|
+
- The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
|
|
41
|
+
hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
|
|
42
|
+
- Cumulative verify means a story's duration includes fixing any regressions it caused.
|
|
43
|
+
|
|
44
|
+
Results live in [`results/`](results/) — one JSON per run, summarized in
|
|
45
|
+
[`RESULTS.md`](RESULTS.md).
|
package/bench/RESULTS.md
CHANGED
|
@@ -1,36 +1,46 @@
|
|
|
1
|
-
# Benchmark results
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
1
|
+
# Benchmark results
|
|
2
|
+
|
|
3
|
+
Result schema v1 records fixture version, sample label, permission profile, telemetry/model
|
|
4
|
+
availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
|
|
5
|
+
|
|
6
|
+
Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
|
|
7
|
+
[README.md](README.md). One row per run — raw JSON in [`results/`](results/).
|
|
8
|
+
|
|
9
|
+
## Runs
|
|
10
|
+
|
|
11
|
+
| Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
|
|
12
|
+
|---|---|---|---|---|---|---|---|---|
|
|
13
|
+
| 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
|
|
14
|
+
| 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
|
|
15
|
+
| 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
|
|
16
|
+
| 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
|
|
17
|
+
| 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
|
|
18
|
+
| — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
|
|
19
|
+
|
|
20
|
+
Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
|
|
21
|
+
passed verify on the first iteration; the final quality check (all 16 assertions on the final
|
|
22
|
+
tree, outside the loop) is green.
|
|
23
|
+
|
|
24
|
+
The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
|
|
25
|
+
before changing the fixture, and the Codex executable could not be probed in this Windows
|
|
26
|
+
environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
|
|
27
|
+
|
|
28
|
+
## What the harness caught before producing a single number
|
|
29
|
+
|
|
30
|
+
Building an honest benchmark is itself a verification pass. The first runs found two real
|
|
31
|
+
Yoke bugs, both fixed in 0.3.0:
|
|
32
|
+
|
|
33
|
+
1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
|
|
34
|
+
5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
|
|
35
|
+
2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
|
|
36
|
+
`-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
|
|
37
|
+
stdin (which selects headless mode) and passes only `--yolo`.
|
|
38
|
+
|
|
39
|
+
## Reading the numbers
|
|
40
|
+
|
|
41
|
+
- Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
|
|
42
|
+
open gap, see the multi-agent design doc).
|
|
43
|
+
- N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
|
|
44
|
+
compare rows.
|
|
45
|
+
- ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
|
|
46
|
+
gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
export const requiredResultFields = [
|
|
2
|
+
'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
|
|
3
|
+
'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
|
|
4
|
+
'iterations', 'finalTestsPass',
|
|
5
|
+
]
|
|
6
|
+
|
|
7
|
+
export function validateResult(result) {
|
|
8
|
+
const missing = requiredResultFields.filter(key => !(key in result))
|
|
9
|
+
if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
|
|
10
|
+
if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
|
|
11
|
+
return result
|
|
12
|
+
}
|