@hecer/yoke 1.3.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +33 -0
  4. package/README.md +108 -16
  5. package/TODOS.md +0 -3
  6. package/bench/README.md +55 -46
  7. package/bench/output-compaction.mjs +65 -0
  8. package/canon/loop/loop-spec.md +22 -8
  9. package/canon/manifest.yaml +1 -1
  10. package/dist/agents/contracts.js +50 -0
  11. package/dist/agents/process-incarnation.js +15 -0
  12. package/dist/agents/process-record.js +65 -0
  13. package/dist/agents/process-streams.js +40 -0
  14. package/dist/agents/process.js +177 -0
  15. package/dist/agents/providers.js +10 -7
  16. package/dist/agents/telemetry.js +62 -0
  17. package/dist/audit/command.js +13 -5
  18. package/dist/cli.js +55 -3
  19. package/dist/loop/candidate-boundaries.js +43 -0
  20. package/dist/loop/candidate-cleanup.js +98 -0
  21. package/dist/loop/candidate-contracts.js +1 -0
  22. package/dist/loop/candidate-selection.js +84 -0
  23. package/dist/loop/candidates.js +228 -0
  24. package/dist/loop/claim-lease.js +131 -0
  25. package/dist/loop/claims.js +177 -40
  26. package/dist/loop/cleanup.js +117 -15
  27. package/dist/loop/decision.js +31 -0
  28. package/dist/loop/dispatcher.js +334 -0
  29. package/dist/loop/git.js +6 -4
  30. package/dist/loop/loop.js +109 -16
  31. package/dist/loop/merge-queue.js +12 -6
  32. package/dist/loop/parallel-adapters.js +185 -0
  33. package/dist/loop/parallel-command.js +287 -0
  34. package/dist/loop/parallel.js +2 -4
  35. package/dist/loop/prd.js +4 -1
  36. package/dist/loop/reporter.js +86 -5
  37. package/dist/loop/run-command.js +216 -58
  38. package/dist/loop/runner.js +67 -32
  39. package/dist/loop/verify.js +65 -11
  40. package/dist/loop/watchdog.js +67 -8
  41. package/dist/loop/worker-cancellation.js +17 -0
  42. package/dist/loop/worker-cleanup.js +23 -0
  43. package/dist/loop/worker-contracts.js +1 -0
  44. package/dist/loop/worker.js +254 -0
  45. package/dist/output/artifact.js +63 -0
  46. package/dist/output/compact.js +192 -0
  47. package/dist/output/types.js +4 -0
  48. package/dist/quality/artifacts.js +59 -0
  49. package/dist/quality/candidate-comparison.js +130 -0
  50. package/dist/quality/command.js +316 -0
  51. package/dist/quality/loop.js +86 -0
  52. package/dist/quality/process-command.js +57 -0
  53. package/dist/quality/reference.js +187 -0
  54. package/dist/quality/repair.js +11 -0
  55. package/dist/quality/runner.js +66 -0
  56. package/dist/quality/types.js +60 -0
  57. package/dist/quality/verdict.js +142 -0
  58. package/dist/retrofit/config.js +26 -2
  59. package/dist/retrofit/gitignore.js +4 -0
  60. package/dist/review/command.js +27 -38
  61. package/dist/review/verdict.js +38 -7
  62. package/docs/MIGRATING-TO-1.4.md +70 -0
  63. package/docs/PUBLISHING.md +16 -2
  64. package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -0
  65. package/docs/superpowers/plans/2026-08-16-artifact-backed-output-compaction.md +329 -0
  66. package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -0
  67. package/docs/superpowers/specs/2026-08-16-artifact-backed-output-compaction-design.md +181 -0
  68. package/gemini-extension.json +1 -1
  69. package/package.json +4 -3
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.3.0",
5
+ "version": "1.5.0",
6
6
  "description": "Cross-agent coding harness: one curated skill canon (TDD, brainstorming, plans, reviews, shipping, design verification) plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.3.0",
3
+ "version": "1.5.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -1,5 +1,38 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.5.0 — 2026-08-16
4
+
5
+ ### Added
6
+ - Failed verify, executable-criterion, performance, configured custom-audit, and completion gates now produce deterministic byte-bounded previews and preserve large complete stdout/stderr in content-addressed `.yoke/artifacts/` files with SHA-256 references.
7
+ - Projects can tune `output.previewBytes` and `output.artifactThresholdBytes`; existing projects use backward-compatible 2 KiB/8 KiB defaults.
8
+ - A deterministic local benchmark verifies signal retention, preview bounds, compression measurement, and artifact digest round-trips without making provider-token claims.
9
+
10
+ ### Security
11
+ - Output artifact paths sanitize story identifiers, stay below a project-local root, use user-only file modes where supported, and are excluded from Yoke's clean-tree and story-commit operations even in upgraded projects. Raw artifacts are never injected automatically and documentation warns that project commands may emit secrets or personal data.
12
+ - Gate command capture is capped at 16 MiB per stdout/stderr stream. Quota overflow fails closed and labels retained evidence as truncated instead of risking unbounded memory or claiming partial output is complete.
13
+
14
+ ## 1.4.0 — 2026-08-15
15
+
16
+ ### Added
17
+ - `yoke loop run --parallel=N` now executes dependency-ready, non-colliding stories through real provider subprocess workers, isolated worktrees, leased claims, and a FIFO integration queue with fresh integrated-system gates.
18
+ - Reference-driven quality declarations can collect screenshots, files, command output, or benchmark results and run a schema-validated blind critic with bounded repair rounds, elapsed-time limits, blocking or advisory policy, and retained proof.
19
+ - `--candidates=N` can fan out up to five isolated implementations, discard mechanically red candidates, select one green candidate through identity-blind pairwise comparison, and preserve selected/loser evidence before cleanup.
20
+ - Loop status now exposes dispatcher, worker, integrator, candidate lifecycle, worktree, queue, integration, reopen, quality-round, repair-budget, and trusted provider/model provenance data.
21
+
22
+ ### Changed
23
+ - Provider subprocesses use explicit lifecycle contracts and incarnation-aware process records so worker cancellation and cleanup target only the process tree Yoke actually started.
24
+ - Parallel and candidate runs disable adaptive routing, honor story-level provider affinity, latch pause requests across the whole dispatcher, and rerun quality plus review after integration.
25
+ - `yoke loop cleanup` retains Yoke worktrees unless `--remove-worktrees` is explicit, while still reaping recorded orphan runners and stale locks safely.
26
+
27
+ ### Fixed
28
+ - Expired claims, worker crashes, merge conflicts, pause races, and integration failures now release ownership deterministically, retain terminal proof, and reopen stories without leaking worktrees or marking false completion.
29
+ - Quality repair fails closed on malformed critic output, reference drift, provider/model provenance mismatch, candidate identity leakage, unavailable critics, exhausted limits, and mechanically red repairs.
30
+ - The watchdog resolves its TypeScript loader from both source and built npm layouts on Node 20+, and read-only Codex comparisons can run in disposable candidate worktrees without weakening normal repository checks.
31
+
32
+ ### Security
33
+ - Blind comparison requests expose only opaque labels and digests while binding every verdict to the trusted judge provider, model, prompt, rubric, reference, and candidate provenance.
34
+ - Cleanup and cancellation use project-scoped leases, owner tokens, PID birth/incarnation checks, and recorded process handles rather than machine-wide process-name matching.
35
+
3
36
  ## 1.3.0 — 2026-08-09
4
37
 
5
38
  ### Added
package/README.md CHANGED
@@ -2,8 +2,8 @@
2
2
 
3
3
  # 🐂 Yoke
4
4
 
5
- <!-- yoke:version:start -->1.3.0<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->657<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.5.0<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->971<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->29<!-- yoke:skills:end -->
8
8
  <!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
9
9
 
@@ -17,7 +17,7 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-657%20passing-brightgreen.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-971%20passing-brightgreen.svg)
21
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
@@ -25,7 +25,16 @@
25
25
 
26
26
  </div>
27
27
 
28
- > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it story by story behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. If any gate is red, nothing is committed. When a story is done, there's a photo of it in `.yoke/proof/<story>/`.
28
+ > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
+
30
+ Yoke 1.5 keeps failed gate output compact without throwing evidence away: deterministic previews
31
+ retain actionable failures and final summaries, while large complete stdout/stderr remains available
32
+ in private, content-addressed local artifacts. Existing projects keep their serial behavior and use
33
+ safe 2 KiB preview / 8 KiB artifact defaults unless configured otherwise.
34
+
35
+ Yoke 1.4 adds opt-in parallel workers and a bounded, reference-driven quality gauntlet without
36
+ changing existing serial loop defaults. See [the 1.4 migration guide](docs/MIGRATING-TO-1.4.md)
37
+ for the new flags, configuration, cleanup behavior, and review-verdict contract.
29
38
 
30
39
  Yoke 1.1 is safe-by-default: provider CLIs use autonomous sandbox profiles unless `--unsafe`
31
40
  is explicit; reviews require a schema-valid verdict and a different model unless
@@ -72,7 +81,7 @@ $ ls reading-app/.yoke/proof/STORY-2/
72
81
  home.png list.png # photographic evidence, labelled per story
73
82
  ```
74
83
 
75
- Every claim in that transcript is enforced by code paths with tests behind them — 657 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
84
+ Every claim in that transcript is enforced by code paths with tests behind them — 971 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
76
85
 
77
86
  ## 🚀 Quickstart
78
87
 
@@ -87,7 +96,7 @@ yoke loop on my-app && yoke loop run my-app --isolate
87
96
  # — or retrofit an existing project —
88
97
  yoke setup /path/to/project # interactive: agents, graph, loop, runner, decisions, routing
89
98
  yoke validate canon # sanity-check the canon
90
- yoke loop run /path/to/project --isolate --reviewer=codex --max=20
99
+ yoke loop run /path/to/project --isolate --parallel=3 --reviewer=codex --max=20
91
100
  ```
92
101
 
93
102
  > Requires Node ≥ 20 and git. No global install? `node /path/to/yoke/dist/cli.js …` or `npm --prefix /path/to/yoke run yoke -- …` work too. The MCP tools (rtk, graphify/Serena, Playwright MCP) are wired by Yoke but installed separately — the generated config is a clearly-labelled, adjustable template.
@@ -156,7 +165,7 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
156
165
  | `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
157
166
  | `yoke change add\|status [dir] [--idea=]` | Queue a change at any time; the loop turns it into append-only stories at the next safe boundary | `0` · `1` invalid inbox/request |
158
167
  | `yoke context init\|status [dir]` | Durable context layer (`PROJECT/DECISIONS/KNOWLEDGE.md`) | `0` |
159
- | `yoke loop on\|off\|status\|decision\|answer\|resume\|run\|cleanup [dir]` | Autonomous loop; `run` is unlimited by default and `--max=N` creates an intentional batch cap; `decision` shows a critical stop, `answer` records it and resumes, `resume` retries a failed restart with the preserved safety options | run: `0` complete · `1` blocked/cap · `2` not runnable / already locked · `3` paused |
168
+ | `yoke loop on\|off\|status\|decision\|answer\|resume\|run\|cleanup [dir]` | Autonomous loop; `run` supports `--parallel=N`, bounded reference-driven `--quality`, and blind `--candidates=N` selection; `--max=N` creates an intentional batch cap; `cleanup` retains worktrees unless `--remove-worktrees` is explicit | run: `0` complete · `1` blocked/cap · `2` not runnable / already locked · `3` paused |
160
169
  | `yoke review [dir] [--reviewer=] [--base=] [--focus=] [--json] [--allow-self-review]` | An independent model writes a schema-valid verdict | `0` approved · `1` findings/invalid verdict · `2` no independent reviewer |
161
170
  | `yoke audit [dir] [--json]` | Dependency, high-confidence secret, and sensitive-change audit | `0` green · `1` blocking findings · `2` not runnable |
162
171
  | `yoke design-scan [dir] [--max=N] [--report]` | Static AI-slop design gate | `0` within budget · `1` over |
@@ -337,6 +346,7 @@ yoke loop run . \
337
346
  --runner=codex \ # implement with Codex…
338
347
  --reviewer=claude \ # …review with Claude (role separation)
339
348
  --isolate \ # each story in a throwaway git worktree
349
+ --parallel=3 \ # run dependency-ready, non-colliding stories concurrently
340
350
  --decision-policy=critical # pause only for high-impact decisions; routine choices stay autonomous
341
351
  # Optional: add --max=20 only when this run should stop after a bounded batch.
342
352
  yoke loop off . # disable
@@ -370,6 +380,49 @@ closes the mechanical false-done paths—targeted evidence, coverage review, cle
370
380
  and integrated journeys—while the project still owns the correctness of its tests and production
371
381
  observability.
372
382
 
383
+ ### Parallel workers and the quality gauntlet
384
+
385
+ `--parallel=N` dispatches dependency-ready stories concurrently. Claims carry leases, workers use
386
+ isolated worktrees, collision areas are serialized, and only a mechanically green candidate enters
387
+ the FIFO integration queue. Integration repeats the project gates against the merged tree; a worker
388
+ success can never bypass a red integrated result. `yoke loop status` reports the dispatcher,
389
+ workers, providers, worktrees, lifecycle, queue, integrations, and reopened stories.
390
+
391
+ Quality is reference-driven and opt-in. Declare what one story should match:
392
+
393
+ ```yaml
394
+ quality:
395
+ reference: { name: approved-home, source: design/home.png, kind: file }
396
+ candidate: { kind: screenshots, paths: [.yoke/proof/STORY-1/home.png] }
397
+ rubric: Match the approved layout, hierarchy, spacing, and states.
398
+ policy: blocking # or advisory
399
+ ```
400
+
401
+ Configure project defaults, then enable the gauntlet for a run:
402
+
403
+ ```yaml
404
+ quality:
405
+ enabled: false # keep opt-in, or make it the project default
406
+ policy: blocking
407
+ maxRounds: 3
408
+ maxMinutes: 60
409
+ consistencyChecks: 2
410
+ maxParallelCandidates: 2
411
+ critic: { agent: codex, model: gpt-5.6-sol } # model required for --candidates
412
+ repair: { agent: claude }
413
+ ```
414
+
415
+ ```bash
416
+ yoke loop run . --quality --quality-rounds=3 --quality-minutes=60
417
+ yoke loop run . --quality --candidates=2 # blind pairwise selection; stories need quality declarations
418
+ ```
419
+
420
+ The critic compares opaque candidate/reference labels, writes schema-validated provenance, and
421
+ cannot modify the project. Blocking findings enter a bounded repair loop and rerun every mechanical
422
+ gate; advisory findings are retained without blocking. `--quality-policy=`, `--no-quality`, and
423
+ `--quality-unbounded` override defaults for one run. Unbounded mode is explicit and warned because
424
+ it removes repair limits, not Yoke's watchdog, isolation, verification, or commit safety.
425
+
373
426
  State lives **outside the model context** — the PRD file plus git — so each iteration is fresh.
374
427
  Use `yoke change add` at any time. Its ignored append-only inbox is consumed at the next story
375
428
  boundary. A separate coverage pass must confirm that every requested outcome maps to behavioral
@@ -393,6 +446,8 @@ Every iteration emits token-free, harness-side feedback (Node console + local fi
393
446
  implementing · iteration 20 · 19/45 (42%) · updated 30s ago
394
447
  ~1h44m remaining (Ø 4m/story)
395
448
  ```
449
+ - **Parallel + quality detail** — active workers include provider, candidate ID, worktree,
450
+ lifecycle, phase, quality round, and repair budget; the integrator is shown separately.
396
451
  - **`.yoke/loop.log`** — an append-only timeline of every phase transition.
397
452
  - **`--json`** — machine mode for supervisors: every status write is *also* emitted as one
398
453
  NDJSON line on stdout (`{"type":"status","state":"running","phase":"verifying",…}` — the
@@ -542,6 +597,41 @@ be fast" is a vibe the loop cannot enforce. Yoke makes it mechanical, at two lev
542
597
  method: profile first, optimize leaves not boundaries, commit benchmarks as tests, version
543
598
  the *why* of every optimization in `context/DECISIONS.md`.
544
599
 
600
+ ### Artifact-backed gate output: compact context, complete local evidence
601
+
602
+ Failed verify, executable-criterion, performance, configured custom-audit, and completion commands can emit thousands of
603
+ low-signal lines. Yoke keeps the model-visible failure summary deterministic and bounded while
604
+ preserving large raw stdout/stderr below `.yoke/artifacts/`:
605
+
606
+ ```yaml
607
+ output:
608
+ previewBytes: 2048 # default: maximum compact preview bytes
609
+ artifactThresholdBytes: 8192 # default: persist raw output only above this size
610
+ ```
611
+
612
+ The preview prioritizes errors, warnings, adjacent context, and final test summaries. Above the
613
+ artifact threshold it also includes a project-relative path, byte count, and full SHA-256 digest,
614
+ for example:
615
+
616
+ ```text
617
+ [full output: .yoke/artifacts/STORY-4/verify-0123abcd4567.log | 42810 bytes | sha256:0123...]
618
+ ```
619
+
620
+ An agent can read that ordinary file when the preview is insufficient; nothing is injected into
621
+ later stories automatically. Repeated identical failures reuse the same content-addressed path.
622
+ Successful gate output is discarded as before. This affects only commands executed by Yoke's own
623
+ gates. It does **not** intercept tool output generated internally by Claude Code, Codex, or Gemini,
624
+ so benchmark ratios for this feature are not provider-token or billing claims.
625
+
626
+ Command capture is capped at 16 MiB per stdout/stderr stream. Exceeding that quota fails the gate
627
+ closed and stores the captured prefix with a `[truncated output: ...]` marker; Yoke never labels
628
+ partial evidence as full output.
629
+
630
+ Yoke treats `.yoke/artifacts/` as local, non-committable runtime state and excludes it from its
631
+ clean-tree and story-commit operations; `yoke retrofit` also adds it to `.gitignore`. Raw command output is intentionally stored
632
+ without redaction so it remains valid evidence and may therefore contain credentials, personal
633
+ data, or other sensitive text emitted by project commands. Inspect artifacts before sharing them.
634
+
545
635
  The loop trusts **verify**, not the agent's exit code: a story whose tests are green is
546
636
  committed even if the agent process exited non-zero (a common Windows `.cmd`-wrapper ghost).
547
637
  A failing verify is retried up to `verify.retries` times (default 1) so a transient flake
@@ -549,7 +639,7 @@ self-heals while a real failure still blocks. Structured acceptance criteria are
549
639
  individually; an unrelated green suite cannot satisfy a criterion without its proof command.
550
640
 
551
641
  `.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, its takeover/recovery leases, lock/decision temp files, `.yoke/story-durations.json`,
552
- `.yoke/ambiguity.md`, and the critical-decision request/answering files are runtime artifacts;
642
+ `.yoke/ambiguity.md`, `.yoke/artifacts/`, and the critical-decision request/answering files are runtime artifacts;
553
643
  `yoke retrofit` gitignores them (along with
554
644
  `.yoke/worktrees/`, `.yoke/backup/`, `.yoke/proof/`, and `.yoke/changes/`) so they never trip the clean-tree gate.
555
645
 
@@ -561,10 +651,11 @@ stale takeover is serialized by `.yoke/loop.lock.takeover`. A second invocation
561
651
  `Another loop is already running here (pid …). If that is wrong, run: yoke loop cleanup`. A lock
562
652
  whose holder process is dead is taken over automatically (with a warning).
563
653
 
564
- **`yoke loop cleanup [dir]`** removes what a crashed loop leaves behind: every worktree under
565
- `.yoke/worktrees/` (via `git worktree remove --force` + `prune` — user-created worktrees are
566
- never touched) and a **stale** lock file. A live lock is reported and left alone. Exits `0`
567
- when everything cleaned, `1` if any removal failed. If a machine/process crash leaves the cleanup
654
+ **`yoke loop cleanup [dir]`** reaps only runner process trees recorded by this project and removes
655
+ a stale lock. Yoke-created worktrees are **retained by default** and listed in the output; pass
656
+ `--remove-worktrees` to remove `.yoke/worktrees/*` with `git worktree remove --force` + `prune`.
657
+ User-created worktrees are never touched. A live lock is reported and left alone. Exits `0` when
658
+ cleanup succeeds, `1` if any requested removal fails. If a machine/process crash leaves the cleanup
568
659
  recovery lease itself behind, an operator can run
569
660
  `yoke loop cleanup . --discard-stale-recovery`; Yoke refuses while its recorded PID is alive, and
570
661
  the force flag must not be run concurrently.
@@ -725,6 +816,7 @@ src/
725
816
  change/ # append-only change inbox · planning · independent coverage review
726
817
  retrofit/ # detect · plan · apply · planners (claude/codex/gemini) · tools
727
818
  loop/ # prd · gates · runner · verify · git/worktree · loop · run-command · lock · cleanup
819
+ quality/ # reference collection · blind critic · bounded repair · candidate comparison
728
820
  new/ # yoke new — greenfield bootstrap
729
821
  prd/ # yoke prd draft|check — idea → stories + lint gate
730
822
  review/ # yoke review — cross-model diff gate
@@ -736,14 +828,14 @@ docs/superpowers/ # the spec and every component's implementation plan
736
828
 
737
829
  ## 🗺️ Roadmap
738
830
 
739
- Yoke 1.1's completed release work moved to the changelog. Remaining, explicitly scoped work
740
- is tracked in [`TODOS.md`](TODOS.md), including provider subprocess wiring for the tested
741
- parallel dispatcher, broader benchmark samples, native output schemas, and release provenance.
831
+ Completed release work lives in the changelog. Remaining, explicitly scoped work is tracked in
832
+ [`TODOS.md`](TODOS.md), including broader benchmark samples, native output schemas, and signed
833
+ release provenance.
742
834
 
743
835
  ## 🧪 Development
744
836
 
745
837
  ```bash
746
- npm test # vitest (657 tests)
838
+ npm test # vitest (971 tests)
747
839
  npm run build # tsc, no emit errors
748
840
  npm run yoke -- validate canon
749
841
  ```
package/TODOS.md CHANGED
@@ -1,8 +1,5 @@
1
1
  # Yoke follow-up work
2
2
 
3
- - Wire the tested async parallel dispatcher to provider subprocess workers. Until then the CLI
4
- rejects `--parallel=N` for `N > 1`; scheduler, claims, and merge queue APIs are available
5
- without claiming a CLI speed-up.
6
3
  - Add provider-native output schemas when all three CLIs expose compatible stable APIs.
7
4
  - Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
8
5
  - Add signed provenance and attestations to npm and GitHub releases.
package/bench/README.md CHANGED
@@ -1,74 +1,83 @@
1
1
  # Yoke benchmark — tokens · speed · quality
2
2
 
3
- Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
- the same PRD within each comparison, and three measured dimensions:
5
-
6
- Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
- multi-run performance study currently measures Codex only; provider support tests are not treated
8
- as performance evidence for Claude or Gemini.
3
+ Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
+ the same PRD within each comparison, and three measured dimensions:
5
+
6
+ Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
+ multi-run performance study currently measures Codex only; provider support tests are not treated
8
+ as performance evidence for Claude or Gemini.
9
9
 
10
10
  | Dimension | How it is measured |
11
11
  |---|---|
12
- | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
12
+ | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
13
13
  | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
14
14
  | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
15
15
 
16
- ## The fixture (`fixtures/string-kit`)
16
+ ## The fixture (`fixtures/string-kit`)
17
17
 
18
18
  A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
19
19
  `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
20
20
  stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
21
21
  final quality check runs everything. No npm installs, so results measure the agent — not the
22
- network.
23
-
24
- ## The routing fixture (`fixtures/routing-queue`)
25
-
26
- A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
- `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
- letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
- enough that a lower-cost worker can potentially repay one routing-controller call per story.
22
+ network.
23
+
24
+ ## The routing fixture (`fixtures/routing-queue`)
25
+
26
+ A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
+ `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
+ letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
+ enough that a lower-cost worker can potentially repay one routing-controller call per story.
30
30
 
31
31
  ## Running it
32
32
 
33
33
  ```bash
34
34
  npm run build
35
- node bench/run.mjs --runner=claude # or gemini / codex
36
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
- node bench/analyze-routing-study.mjs
43
- node bench/run-matrix.mjs --label=release-1.0
35
+ node bench/run.mjs --runner=claude # or gemini / codex
36
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
+ node bench/analyze-routing-study.mjs
43
+ node bench/run-matrix.mjs --label=release-1.0
44
+ node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
44
45
  ```
45
46
 
46
- Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
47
- explicit `--run-root`), git-inits it,
48
- drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
47
+ Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
48
+ explicit `--run-root`), git-inits it,
49
+ drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
49
50
  `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
50
51
  The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
51
- failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
52
-
53
- `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
54
- routing registry, junctions this checkout's dependencies, and replays the seed's original
55
- acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
56
- visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
57
- to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
58
- aggregates the checked-in three-pair Codex-only study.
52
+ failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
53
+
54
+ ### Gate-output compaction benchmark
55
+
56
+ `output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
57
+ It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
58
+ remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
59
+ bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
60
+ comparison to Aphrodite's corpus or published ratios.
61
+
62
+ `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
63
+ routing registry, junctions this checkout's dependencies, and replays the seed's original
64
+ acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
65
+ visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
66
+ to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
67
+ aggregates the checked-in three-pair Codex-only study.
59
68
 
60
69
  ## Caveats (read before quoting numbers)
61
70
 
62
- - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
63
- alternating-order pairs per arm. Re-run before setting broad policy defaults.
64
- - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
65
- policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
66
- to exclude personal MCP/plugin startup from both sides.
67
- - Model identity matters more than CLI identity: `tokens.model` records what actually served
68
- the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
69
- - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
70
- treat cached input as equivalent to newly processed input. Dollar cost is reported only when
71
- the provider emits it—Yoke does not guess prices from a model name.
71
+ - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
72
+ alternating-order pairs per arm. Re-run before setting broad policy defaults.
73
+ - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
74
+ policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
75
+ to exclude personal MCP/plugin startup from both sides.
76
+ - Model identity matters more than CLI identity: `tokens.model` records what actually served
77
+ the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
78
+ - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
79
+ treat cached input as equivalent to newly processed input. Dollar cost is reported only when
80
+ the provider emits it—Yoke does not guess prices from a model name.
72
81
  - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
73
82
  hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
74
83
  - Cumulative verify means a story's duration includes fixing any regressions it caused.
@@ -0,0 +1,65 @@
1
+ import { createHash } from 'node:crypto'
2
+ import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
3
+ import { tmpdir } from 'node:os'
4
+ import { join, resolve } from 'node:path'
5
+ import { fileURLToPath } from 'node:url'
6
+
7
+ async function runtimeModules() {
8
+ const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
9
+ const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
10
+ const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
11
+ return Promise.all([
12
+ import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
13
+ import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
14
+ ])
15
+ }
16
+
17
+ function fixture() {
18
+ return [
19
+ '=== stdout ===',
20
+ 'compiling application',
21
+ 'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
22
+ 'the failing call is part of the checkout flow',
23
+ ...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
24
+ '=== stderr ===',
25
+ 'Tests: 1 failed, 249 passed, 250 total',
26
+ ].join('\n')
27
+ }
28
+
29
+ export async function runOutputCompactionBenchmark() {
30
+ const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
31
+ const raw = fixture()
32
+ const previewBudgetBytes = 512
33
+ const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
34
+ const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
35
+ try {
36
+ const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
37
+ const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
38
+ const storedDigest = createHash('sha256').update(stored).digest('hex')
39
+ const previewBytes = Buffer.byteLength(compacted.preview)
40
+ const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
41
+ return {
42
+ fixture: 'gate-output-v1',
43
+ rawBytes: Buffer.byteLength(raw),
44
+ rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
45
+ previewBudgetBytes,
46
+ previewBytes,
47
+ previewApproxTokens: Math.ceil(previewBytes / 4),
48
+ referencedBytes,
49
+ compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
50
+ earlyErrorRetained: compacted.preview.includes('error TS2304'),
51
+ finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
52
+ digestRoundTrip: storedDigest === artifact.sha256,
53
+ }
54
+ } finally {
55
+ rmSync(dir, { recursive: true, force: true })
56
+ }
57
+ }
58
+
59
+ if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
60
+ const result = await runOutputCompactionBenchmark()
61
+ console.log(JSON.stringify(result, null, 2))
62
+ if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
63
+ process.exitCode = 1
64
+ }
65
+ }
@@ -4,13 +4,21 @@ The autonomous loop is optional and toggle-able:
4
4
 
5
5
  - `yoke loop on` / `yoke loop off` — enable or disable it in `.yoke/config.yaml`.
6
6
  - `yoke loop status` — show enabled state and backlog progress.
7
- - `yoke loop run [--max=N] [--isolate] [--decision-policy=auto|critical]` — run until the current backlog is green or a gate blocks.
7
+ - `yoke loop run [--max=N] [--parallel=N] [--isolate] [--decision-policy=auto|critical] [--quality|--no-quality] [--quality-rounds=N] [--quality-minutes=N] [--quality-policy=blocking|advisory] [--quality-unbounded] [--candidates=N]` — run until the current backlog is green or a gate blocks.
8
8
  - `yoke change add --idea="..."` — queue a product change at any time, including while the loop is running.
9
9
  - `yoke loop decision` / `yoke loop answer --choice=<id>` — inspect and answer a structured critical stop.
10
10
 
11
11
  Pass `--isolate` to implement each story in a fresh git worktree. Only a verified, committed
12
12
  story is fast-forwarded to the main tree. Pass `--review` or `--reviewer=<provider>` to require
13
- a separate, schema-validated review. Pass `--json` for NDJSON status on stdout.
13
+ a separate, schema-validated review. Pass `--parallel=N` to dispatch ready, non-colliding stories
14
+ concurrently. Pass `--json` for NDJSON status on stdout.
15
+
16
+ Stories may declare a reference, candidate artifact, rubric, and blocking/advisory quality policy.
17
+ `--quality` runs a read-only blind critic plus bounded repair before review; every repair reruns the
18
+ mechanical gates. `--candidates=N` requires quality declarations and dispatches multiple isolated
19
+ implementations, rejects mechanically red candidates, selects one green candidate through opaque
20
+ pairwise handles, and retains every candidate's terminal proof before cleanup. Parallel/candidate
21
+ runs do not combine with adaptive routing.
14
22
 
15
23
  At every story boundary, Yoke consumes at most one queued change. The configured Claude,
16
24
  Codex, or Gemini provider may propose only new stories in a separate runtime file. A fresh
@@ -21,7 +29,8 @@ stories untouched. The request stays pending on any failure or uncovered outcome
21
29
  For each story:
22
30
 
23
31
  1. Require a clean git worktree.
24
- 2. Pick the highest-priority ready unfinished story.
32
+ 2. Pick the highest-priority ready unfinished story, or claim multiple dependency-ready stories
33
+ whose collision areas do not overlap when parallel dispatch is enabled.
25
34
  3. Stop the line if acceptance is empty. With `verify.requireCriteria: true`, every criterion
26
35
  must be structured. Every structured criterion, including in compatible legacy projects,
27
36
  must use a single approved test command containing its criterion ID and no shell operators.
@@ -30,16 +39,21 @@ For each story:
30
39
  material cost, compliance, or irreversible choices may pause for a human decision.
31
40
  5. Run every structured criterion's targeted commands and write
32
41
  `.yoke/proof/<story>/evidence.json`. Then run project-wide `verify.command` (or detected
33
- `npm test`). Performance, audit, and independent review gates follow when configured. Any
34
- failure blocks, and no proof command runs after review.
35
- 6. Only after all gates pass, mark the story `passes: true`, log the decision, and commit
42
+ `npm test`). Performance and audit follow when configured. If quality is enabled, collect the
43
+ declared artifact, run the blind critic, and repair within the configured round/time bounds.
44
+ Independent review follows. Any blocking failure stops the candidate.
45
+ 6. Parallel workers enqueue green candidate commits. The integrator applies one at a time and
46
+ reruns mechanical gates, fresh quality, and review against the integrated tree. Failed
47
+ integration reopens the story and retains proof.
48
+ 7. Only after all gates pass, mark the story `passes: true`, log the decision, and commit
36
49
  atomically. A failed commit restores the PRD state.
37
- 7. When all current stories pass, run optional `completion.command` against the integrated
50
+ 8. When all current stories pass, run optional `completion.command` against the integrated
38
51
  system. Only a green result reports `complete`; otherwise the loop blocks. This readiness
39
52
  result is ephemeral, not a release and not a freeze on future changes.
40
53
 
41
54
  A supervisor can pause the loop by creating `.yoke/loop.pause`. The running story finishes;
42
- the signal is consumed at the next story boundary and the process exits with code `3`.
55
+ the dispatcher latches the signal, stops launching new workers, lets active workers reach safe
56
+ terminal proof/cleanup, and exits with code `3` before another story is integrated.
43
57
 
44
58
  State lives outside model context: PRD, git, and the ignored `.yoke/changes/` inbox. All are
45
59
  re-read at story boundaries, so a request queued mid-run becomes additional stories without a
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.2.0
2
+ version: 1.5.0
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology }
@@ -0,0 +1,50 @@
1
+ import { z } from 'zod';
2
+ export const AgentSchema = z.enum(['claude', 'codex', 'gemini']);
3
+ export const PermissionProfileSchema = z.enum(['safe', 'unsafe', 'read-only']);
4
+ export const ModelSelectionSchema = z.object({
5
+ model: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/).optional(),
6
+ reasoningEffort: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$/).optional(),
7
+ nativeMultiAgent: z.boolean().optional(),
8
+ bare: z.boolean().optional(),
9
+ });
10
+ export const AgentInvocationSchema = z.object({
11
+ command: z.string().min(1),
12
+ args: z.array(z.string()),
13
+ input: z.string(),
14
+ cwd: z.string().min(1),
15
+ });
16
+ const ProviderTokenUsageSchema = z.object({
17
+ inputTokens: z.number().nonnegative(),
18
+ cachedInputTokens: z.number().nonnegative().optional(),
19
+ cacheWriteInputTokens: z.number().nonnegative().optional(),
20
+ outputTokens: z.number().nonnegative(),
21
+ reasoningOutputTokens: z.number().nonnegative().optional(),
22
+ totalCostUsd: z.number().nonnegative().optional(),
23
+ model: z.string().min(1).optional(),
24
+ });
25
+ export const ProviderTelemetrySchema = z.object({
26
+ usageAvailable: z.boolean(),
27
+ tokens: ProviderTokenUsageSchema.optional(),
28
+ }).superRefine((telemetry, ctx) => {
29
+ if (telemetry.usageAvailable && !telemetry.tokens) {
30
+ ctx.addIssue({ code: 'custom', path: ['tokens'], message: 'usageAvailable telemetry requires token totals' });
31
+ }
32
+ });
33
+ const MachineRoleSchema = z.enum([
34
+ 'route',
35
+ 'review',
36
+ 'quality',
37
+ 'decomposition',
38
+ 'candidate-selection',
39
+ 'telemetry',
40
+ ]);
41
+ export const MachineEnvelopeSchema = z.object({
42
+ schemaVersion: z.literal(1),
43
+ provider: AgentSchema,
44
+ model: z.string().min(1).optional(),
45
+ role: MachineRoleSchema,
46
+ durationMs: z.number().int().nonnegative(),
47
+ permissions: PermissionProfileSchema,
48
+ usage: ProviderTokenUsageSchema.optional(),
49
+ raw: z.record(z.unknown()).optional(),
50
+ });
@@ -0,0 +1,15 @@
1
+ import { execFileSync } from 'node:child_process';
2
+ const queryProcessIdentity = (command, args, options) => execFileSync(command, args, { stdio: 'pipe', ...options }).toString();
3
+ export function processIncarnation(pid, platform = process.platform, query = queryProcessIdentity) {
4
+ try {
5
+ if (platform === 'win32') {
6
+ const output = query('powershell.exe', ['-NoProfile', '-NonInteractive', '-Command', '(Get-CimInstance Win32_Process -Filter "ProcessId = $env:YOKE_PROCESS_PID").CreationDate'], { env: { ...process.env, YOKE_PROCESS_PID: String(pid) } }).trim();
7
+ return output ? `win32:${output}` : undefined;
8
+ }
9
+ const output = query('ps', ['-o', 'lstart=', '-p', String(pid)]).trim();
10
+ return output ? `posix:${output}` : undefined;
11
+ }
12
+ catch {
13
+ return undefined;
14
+ }
15
+ }