@hecer/yoke 1.4.0 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.4.0",
5
+ "version": "1.5.0",
6
6
  "description": "Cross-agent coding harness: one curated skill canon (TDD, brainstorming, plans, reviews, shipping, design verification) plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.4.0",
3
+ "version": "1.5.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -1,6 +1,24 @@
1
1
  # Changelog
2
2
 
3
- ## 1.4.0 — 2026-08-15
3
+ ## Unreleased
4
+
5
+ ## 1.5.1 — 2026-08-17
6
+
7
+ ### Fixed
8
+ - Serena MCP configurations generated by Yoke no longer open the local web dashboard automatically on every startup.
9
+
10
+ ## 1.5.0 — 2026-08-16
11
+
12
+ ### Added
13
+ - Failed verify, executable-criterion, performance, configured custom-audit, and completion gates now produce deterministic byte-bounded previews and preserve large complete stdout/stderr in content-addressed `.yoke/artifacts/` files with SHA-256 references.
14
+ - Projects can tune `output.previewBytes` and `output.artifactThresholdBytes`; existing projects use backward-compatible 2 KiB/8 KiB defaults.
15
+ - A deterministic local benchmark verifies signal retention, preview bounds, compression measurement, and artifact digest round-trips without making provider-token claims.
16
+
17
+ ### Security
18
+ - Output artifact paths sanitize story identifiers, stay below a project-local root, use user-only file modes where supported, and are excluded from Yoke's clean-tree and story-commit operations even in upgraded projects. Raw artifacts are never injected automatically and documentation warns that project commands may emit secrets or personal data.
19
+ - Gate command capture is capped at 16 MiB per stdout/stderr stream. Quota overflow fails closed and labels retained evidence as truncated instead of risking unbounded memory or claiming partial output is complete.
20
+
21
+ ## 1.4.0 — 2026-08-15
4
22
 
5
23
  ### Added
6
24
  - `yoke loop run --parallel=N` now executes dependency-ready, non-colliding stories through real provider subprocess workers, isolated worktrees, leased claims, and a FIFO integration queue with fresh integrated-system gates.
package/README.md CHANGED
@@ -2,8 +2,8 @@
2
2
 
3
3
  # 🐂 Yoke
4
4
 
5
- <!-- yoke:version:start -->1.4.0<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->928<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.5.1<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->971<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->29<!-- yoke:skills:end -->
8
8
  <!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
9
9
 
@@ -17,7 +17,7 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-928%20passing-brightgreen.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-971%20passing-brightgreen.svg)
21
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
@@ -25,9 +25,14 @@
25
25
 
26
26
  </div>
27
27
 
28
- > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
-
30
- Yoke 1.4 adds opt-in parallel workers and a bounded, reference-driven quality gauntlet without
28
+ > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
+
30
+ Yoke 1.5 keeps failed gate output compact without throwing evidence away: deterministic previews
31
+ retain actionable failures and final summaries, while large complete stdout/stderr remains available
32
+ in private, content-addressed local artifacts. Existing projects keep their serial behavior and use
33
+ safe 2 KiB preview / 8 KiB artifact defaults unless configured otherwise.
34
+
35
+ Yoke 1.4 adds opt-in parallel workers and a bounded, reference-driven quality gauntlet without
31
36
  changing existing serial loop defaults. See [the 1.4 migration guide](docs/MIGRATING-TO-1.4.md)
32
37
  for the new flags, configuration, cleanup behavior, and review-verdict contract.
33
38
 
@@ -76,7 +81,7 @@ $ ls reading-app/.yoke/proof/STORY-2/
76
81
  home.png list.png # photographic evidence, labelled per story
77
82
  ```
78
83
 
79
- Every claim in that transcript is enforced by code paths with tests behind them — 928 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
84
+ Every claim in that transcript is enforced by code paths with tests behind them — 971 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
80
85
 
81
86
  ## 🚀 Quickstart
82
87
 
@@ -592,6 +597,41 @@ be fast" is a vibe the loop cannot enforce. Yoke makes it mechanical, at two lev
592
597
  method: profile first, optimize leaves not boundaries, commit benchmarks as tests, version
593
598
  the *why* of every optimization in `context/DECISIONS.md`.
594
599
 
600
+ ### Artifact-backed gate output: compact context, complete local evidence
601
+
602
+ Failed verify, executable-criterion, performance, configured custom-audit, and completion commands can emit thousands of
603
+ low-signal lines. Yoke keeps the model-visible failure summary deterministic and bounded while
604
+ preserving large raw stdout/stderr below `.yoke/artifacts/`:
605
+
606
+ ```yaml
607
+ output:
608
+ previewBytes: 2048 # default: maximum compact preview bytes
609
+ artifactThresholdBytes: 8192 # default: persist raw output only above this size
610
+ ```
611
+
612
+ The preview prioritizes errors, warnings, adjacent context, and final test summaries. Above the
613
+ artifact threshold it also includes a project-relative path, byte count, and full SHA-256 digest,
614
+ for example:
615
+
616
+ ```text
617
+ [full output: .yoke/artifacts/STORY-4/verify-0123abcd4567.log | 42810 bytes | sha256:0123...]
618
+ ```
619
+
620
+ An agent can read that ordinary file when the preview is insufficient; nothing is injected into
621
+ later stories automatically. Repeated identical failures reuse the same content-addressed path.
622
+ Successful gate output is discarded as before. This affects only commands executed by Yoke's own
623
+ gates. It does **not** intercept tool output generated internally by Claude Code, Codex, or Gemini,
624
+ so benchmark ratios for this feature are not provider-token or billing claims.
625
+
626
+ Command capture is capped at 16 MiB per stdout/stderr stream. Exceeding that quota fails the gate
627
+ closed and stores the captured prefix with a `[truncated output: ...]` marker; Yoke never labels
628
+ partial evidence as full output.
629
+
630
+ Yoke treats `.yoke/artifacts/` as local, non-committable runtime state and excludes it from its
631
+ clean-tree and story-commit operations; `yoke retrofit` also adds it to `.gitignore`. Raw command output is intentionally stored
632
+ without redaction so it remains valid evidence and may therefore contain credentials, personal
633
+ data, or other sensitive text emitted by project commands. Inspect artifacts before sharing them.
634
+
595
635
  The loop trusts **verify**, not the agent's exit code: a story whose tests are green is
596
636
  committed even if the agent process exited non-zero (a common Windows `.cmd`-wrapper ghost).
597
637
  A failing verify is retried up to `verify.retries` times (default 1) so a transient flake
@@ -599,7 +639,7 @@ self-heals while a real failure still blocks. Structured acceptance criteria are
599
639
  individually; an unrelated green suite cannot satisfy a criterion without its proof command.
600
640
 
601
641
  `.yoke/loop-status.json`, `.yoke/loop.log`, `.yoke/loop.lock`, its takeover/recovery leases, lock/decision temp files, `.yoke/story-durations.json`,
602
- `.yoke/ambiguity.md`, and the critical-decision request/answering files are runtime artifacts;
642
+ `.yoke/ambiguity.md`, `.yoke/artifacts/`, and the critical-decision request/answering files are runtime artifacts;
603
643
  `yoke retrofit` gitignores them (along with
604
644
  `.yoke/worktrees/`, `.yoke/backup/`, `.yoke/proof/`, and `.yoke/changes/`) so they never trip the clean-tree gate.
605
645
 
@@ -795,7 +835,7 @@ release provenance.
795
835
  ## 🧪 Development
796
836
 
797
837
  ```bash
798
- npm test # vitest (928 tests)
838
+ npm test # vitest (971 tests)
799
839
  npm run build # tsc, no emit errors
800
840
  npm run yoke -- validate canon
801
841
  ```
package/bench/README.md CHANGED
@@ -1,74 +1,83 @@
1
1
  # Yoke benchmark — tokens · speed · quality
2
2
 
3
- Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
- the same PRD within each comparison, and three measured dimensions:
5
-
6
- Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
- multi-run performance study currently measures Codex only; provider support tests are not treated
8
- as performance evidence for Claude or Gemini.
3
+ Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
4
+ the same PRD within each comparison, and three measured dimensions:
5
+
6
+ Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
7
+ multi-run performance study currently measures Codex only; provider support tests are not treated
8
+ as performance evidence for Claude or Gemini.
9
9
 
10
10
  | Dimension | How it is measured |
11
11
  |---|---|
12
- | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
12
+ | **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
13
13
  | **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
14
14
  | **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
15
15
 
16
- ## The fixture (`fixtures/string-kit`)
16
+ ## The fixture (`fixtures/string-kit`)
17
17
 
18
18
  A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
19
19
  `node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
20
20
  stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
21
21
  final quality check runs everything. No npm installs, so results measure the agent — not the
22
- network.
23
-
24
- ## The routing fixture (`fixtures/routing-queue`)
25
-
26
- A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
- `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
- letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
- enough that a lower-cost worker can potentially repay one routing-controller call per story.
22
+ network.
23
+
24
+ ## The routing fixture (`fixtures/routing-queue`)
25
+
26
+ A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
27
+ `node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
28
+ letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
29
+ enough that a lower-cost worker can potentially repay one routing-controller call per story.
30
30
 
31
31
  ## Running it
32
32
 
33
33
  ```bash
34
34
  npm run build
35
- node bench/run.mjs --runner=claude # or gemini / codex
36
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
- node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
- node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
- node bench/analyze-routing-study.mjs
43
- node bench/run-matrix.mjs --label=release-1.0
35
+ node bench/run.mjs --runner=claude # or gemini / codex
36
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
37
+ node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
38
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
39
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
40
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
41
+ node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
42
+ node bench/analyze-routing-study.mjs
43
+ node bench/run-matrix.mjs --label=release-1.0
44
+ node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
44
45
  ```
45
46
 
46
- Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
47
- explicit `--run-root`), git-inits it,
48
- drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
47
+ Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
48
+ explicit `--run-root`), git-inits it,
49
+ drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
49
50
  `bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
50
51
  The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
51
- failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
52
-
53
- `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
54
- routing registry, junctions this checkout's dependencies, and replays the seed's original
55
- acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
56
- visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
57
- to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
58
- aggregates the checked-in three-pair Codex-only study.
52
+ failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
53
+
54
+ ### Gate-output compaction benchmark
55
+
56
+ `output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
57
+ It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
58
+ remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
59
+ bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
60
+ comparison to Aphrodite's corpus or published ratios.
61
+
62
+ `run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
63
+ routing registry, junctions this checkout's dependencies, and replays the seed's original
64
+ acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
65
+ visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
66
+ to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
67
+ aggregates the checked-in three-pair Codex-only study.
59
68
 
60
69
  ## Caveats (read before quoting numbers)
61
70
 
62
- - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
63
- alternating-order pairs per arm. Re-run before setting broad policy defaults.
64
- - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
65
- policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
66
- to exclude personal MCP/plugin startup from both sides.
67
- - Model identity matters more than CLI identity: `tokens.model` records what actually served
68
- the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
69
- - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
70
- treat cached input as equivalent to newly processed input. Dollar cost is reported only when
71
- the provider emits it—Yoke does not guess prices from a model name.
71
+ - Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
72
+ alternating-order pairs per arm. Re-run before setting broad policy defaults.
73
+ - Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
74
+ policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
75
+ to exclude personal MCP/plugin startup from both sides.
76
+ - Model identity matters more than CLI identity: `tokens.model` records what actually served
77
+ the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
78
+ - Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
79
+ treat cached input as equivalent to newly processed input. Dollar cost is reported only when
80
+ the provider emits it—Yoke does not guess prices from a model name.
72
81
  - The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
73
82
  hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
74
83
  - Cumulative verify means a story's duration includes fixing any regressions it caused.
@@ -0,0 +1,65 @@
1
+ import { createHash } from 'node:crypto'
2
+ import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
3
+ import { tmpdir } from 'node:os'
4
+ import { join, resolve } from 'node:path'
5
+ import { fileURLToPath } from 'node:url'
6
+
7
+ async function runtimeModules() {
8
+ const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
9
+ const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
10
+ const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
11
+ return Promise.all([
12
+ import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
13
+ import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
14
+ ])
15
+ }
16
+
17
+ function fixture() {
18
+ return [
19
+ '=== stdout ===',
20
+ 'compiling application',
21
+ 'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
22
+ 'the failing call is part of the checkout flow',
23
+ ...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
24
+ '=== stderr ===',
25
+ 'Tests: 1 failed, 249 passed, 250 total',
26
+ ].join('\n')
27
+ }
28
+
29
+ export async function runOutputCompactionBenchmark() {
30
+ const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
31
+ const raw = fixture()
32
+ const previewBudgetBytes = 512
33
+ const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
34
+ const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
35
+ try {
36
+ const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
37
+ const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
38
+ const storedDigest = createHash('sha256').update(stored).digest('hex')
39
+ const previewBytes = Buffer.byteLength(compacted.preview)
40
+ const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
41
+ return {
42
+ fixture: 'gate-output-v1',
43
+ rawBytes: Buffer.byteLength(raw),
44
+ rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
45
+ previewBudgetBytes,
46
+ previewBytes,
47
+ previewApproxTokens: Math.ceil(previewBytes / 4),
48
+ referencedBytes,
49
+ compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
50
+ earlyErrorRetained: compacted.preview.includes('error TS2304'),
51
+ finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
52
+ digestRoundTrip: storedDigest === artifact.sha256,
53
+ }
54
+ } finally {
55
+ rmSync(dir, { recursive: true, force: true })
56
+ }
57
+ }
58
+
59
+ if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
60
+ const result = await runOutputCompactionBenchmark()
61
+ console.log(JSON.stringify(result, null, 2))
62
+ if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
63
+ process.exitCode = 1
64
+ }
65
+ }
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.4.0
2
+ version: 1.5.1
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology }
@@ -4,4 +4,8 @@ MIT, MCP-first. The alternative to graphify, selected via `yoke retrofit --code-
4
4
 
5
5
  Wired as an MCP server for all three agents. Best for large, strongly-typed codebases (TypeScript, Python, Go) doing systematic refactoring, where missing a caller is costly.
6
6
 
7
- Caveat: needs one language server per language (can be fiddly on Windows for exotic languages) and requires `uv`. The launch command is a best-effort template — adjust to your install, e.g. `uvx --from git+https://github.com/oraios/serena serena-mcp-server`.
7
+ Caveat: needs one language server per language (can be fiddly on Windows for exotic languages) and requires `uv`. The launch command is a best-effort template — adjust to your install, e.g. `uvx --from git+https://github.com/oraios/serena serena-mcp-server`.
8
+
9
+ Yoke disables Serena's automatic web-dashboard launch in generated MCP configurations. The
10
+ dashboard server remains available for manual inspection, but selecting Serena no longer opens a
11
+ browser window on every MCP startup.
@@ -22,12 +22,20 @@ export function runAudit(targetDir, opts = {}) {
22
22
  }
23
23
  findings.push(...scanSensitiveChanges(changed));
24
24
  if (opts.command) {
25
- try {
26
- execSync(opts.command, { cwd: targetDir, stdio: 'pipe' });
27
- }
28
- catch {
29
- findings.push({ ruleId: 'audit.custom-command', severity: 'high', message: `Audit command failed: ${opts.command}`, file: '.yoke/config.yaml' });
25
+ let commandResult;
26
+ if (opts.commandRunner)
27
+ commandResult = opts.commandRunner(opts.command, targetDir);
28
+ else {
29
+ try {
30
+ execSync(opts.command, { cwd: targetDir, stdio: 'pipe' });
31
+ commandResult = { passed: true, summary: `audit passed: ${opts.command}` };
32
+ }
33
+ catch {
34
+ commandResult = { passed: false, summary: `Audit command failed: ${opts.command}` };
35
+ }
30
36
  }
37
+ if (!commandResult.passed)
38
+ findings.push({ ruleId: 'audit.custom-command', severity: 'high', message: commandResult.summary, file: '.yoke/config.yaml' });
31
39
  }
32
40
  else {
33
41
  const dependency = opts.dependency ?? ((repoFiles) => {
package/dist/loop/git.js CHANGED
@@ -1,14 +1,16 @@
1
1
  import { execFileSync } from 'node:child_process';
2
2
  import { sanitizeCommitMessage } from './identity.js';
3
+ const OUTPUT_ARTIFACT_EXCLUDE = ':(exclude).yoke/artifacts/**';
3
4
  export const realGitOps = {
4
5
  isClean(dir) {
5
- const out = execFileSync('git', ['status', '--porcelain'], { cwd: dir }).toString();
6
+ const out = execFileSync('git', ['status', '--porcelain', '--untracked-files=all', '--', '.', OUTPUT_ARTIFACT_EXCLUDE], { cwd: dir }).toString();
6
7
  return out.trim() === '';
7
8
  },
8
9
  commitAll(dir, message, identity) {
9
- execFileSync('git', ['add', '-A'], { cwd: dir, stdio: 'pipe' });
10
- const status = execFileSync('git', ['status', '--porcelain'], { cwd: dir }).toString().trim();
11
- if (status === '') {
10
+ execFileSync('git', ['reset', '--quiet', '--', '.yoke/artifacts'], { cwd: dir, stdio: 'pipe' });
11
+ execFileSync('git', ['add', '-A', '--', '.', OUTPUT_ARTIFACT_EXCLUDE], { cwd: dir, stdio: 'pipe' });
12
+ const staged = execFileSync('git', ['diff', '--cached', '--name-only'], { cwd: dir }).toString().trim();
13
+ if (staged === '') {
12
14
  throw new Error('nothing to commit after agent run');
13
15
  }
14
16
  const identityArgs = identity
@@ -1,6 +1,6 @@
1
1
  import { join } from 'node:path';
2
2
  import { existsSync } from 'node:fs';
3
- import { loadConfig, saveConfig, defaultConfig, resolveVerifyCommand } from '../retrofit/config.js';
3
+ import { loadConfig, saveConfig, defaultConfig, resolveOutputPolicy, resolveVerifyCommand } from '../retrofit/config.js';
4
4
  import { loadPrd, progress } from './prd.js';
5
5
  import { runLoop } from './loop.js';
6
6
  import { commitPaths, realGitOps } from './git.js';
@@ -144,6 +144,7 @@ export function runLoopCommand(targetDir, opts) {
144
144
  return 2;
145
145
  }
146
146
  }
147
+ const outputPolicy = resolveOutputPolicy(config);
147
148
  let verify = opts.verify;
148
149
  if (!verify) {
149
150
  const command = resolveVerifyCommand(targetDir, config);
@@ -151,16 +152,16 @@ export function runLoopCommand(targetDir, opts) {
151
152
  console.error('No verify command configured. Set verify.command in .yoke/config.yaml (e.g. "npm test") so the loop can confirm tests pass before marking work done.');
152
153
  return 2;
153
154
  }
154
- verify = retryingVerifier(commandVerifier(command), config.verify?.retries ?? 1);
155
+ verify = retryingVerifier(commandVerifier(command, { phase: 'verify', policy: outputPolicy }), config.verify?.retries ?? 1);
155
156
  }
156
157
  // Optional performance budget gate: same contract as verify (exit 0 = within
157
158
  // budget), same flake tolerance (benchmarks are noisy).
158
159
  let perf = opts.perf;
159
160
  if (!perf && config.perf?.command) {
160
- perf = retryingVerifier(commandVerifier(config.perf.command), config.perf.retries ?? 1);
161
+ perf = retryingVerifier(commandVerifier(config.perf.command, { phase: 'perf', policy: outputPolicy }), config.perf.retries ?? 1);
161
162
  }
162
163
  const completion = config.completion?.command
163
- ? retryingVerifier(commandVerifier(config.completion.command), config.completion.retries ?? 1)
164
+ ? retryingVerifier(commandVerifier(config.completion.command, { phase: 'completion', policy: outputPolicy }), config.completion.retries ?? 1)
164
165
  : undefined;
165
166
  // Opt-in self-update, loop START only — this run keeps executing the version
166
167
  // it started with; a fetched upgrade applies from the next invocation.
@@ -181,8 +182,12 @@ export function runLoopCommand(targetDir, opts) {
181
182
  let audit = opts.audit;
182
183
  if (!audit && config.audit?.enabled) {
183
184
  audit = (dir) => {
184
- const result = runAudit(dir, { command: config.audit?.command, suppressions: config.audit?.suppressions });
185
- return { passed: result.code === 0, summary: result.error ?? (result.findings.map(f => `${f.ruleId} ${f.file}${f.line ? `:${f.line}` : ''}`).join(', ') || 'audit passed') };
185
+ const result = runAudit(dir, {
186
+ command: config.audit?.command,
187
+ suppressions: config.audit?.suppressions,
188
+ commandRunner: (command, commandDir) => commandVerifier(command, { phase: 'audit', policy: outputPolicy })(commandDir),
189
+ });
190
+ return { passed: result.code === 0, summary: result.error ?? (result.findings.map(f => `${f.ruleId} ${f.file}${f.line ? `:${f.line}` : ''}: ${f.message}`).join('\n') || 'audit passed') };
186
191
  };
187
192
  }
188
193
  if (commitIdentity) {
@@ -413,7 +418,7 @@ export function runLoopCommand(targetDir, opts) {
413
418
  git: opts.git,
414
419
  identity: commitIdentity,
415
420
  verify,
416
- verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify)(dir),
421
+ verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify, { phase: 'criterion', policy: outputPolicy })(dir),
417
422
  requireCriterionEvidence: config.verify?.requireCriteria ?? false,
418
423
  perf,
419
424
  audit,
@@ -433,7 +438,7 @@ export function runLoopCommand(targetDir, opts) {
433
438
  git,
434
439
  commitIdentity,
435
440
  verify,
436
- verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify)(dir),
441
+ verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify, { phase: 'criterion', policy: outputPolicy })(dir),
437
442
  requireCriterionEvidence: config.verify?.requireCriteria ?? false,
438
443
  completion,
439
444
  intake,
@@ -1,26 +1,80 @@
1
1
  import { execSync } from 'node:child_process';
2
- // Runs a shell command in the target dir; passed = exit 0. execSync goes through the
3
- // shell, so `npm test` resolves npm.cmd on Windows. Output is captured (not streamed).
4
- export function commandVerifier(command) {
2
+ import { compactCommandOutput } from '../output/compact.js';
3
+ import { writeOutputArtifact } from '../output/artifact.js';
4
+ import { DEFAULT_OUTPUT_POLICY } from '../output/types.js';
5
+ function outputText(value) {
6
+ if (Buffer.isBuffer(value))
7
+ return value.toString('utf8');
8
+ return typeof value === 'string' ? value : '';
9
+ }
10
+ function labelledOutput(stdout, stderr) {
11
+ const sections = [];
12
+ if (stdout)
13
+ sections.push(`=== stdout ===\n${stdout}`);
14
+ if (stderr)
15
+ sections.push(`=== stderr ===\n${stderr}`);
16
+ return sections.join(stdout.endsWith('\n') ? '' : '\n');
17
+ }
18
+ function errorMessage(error) {
19
+ return (error instanceof Error ? error.message : String(error))
20
+ .replace(/[\u0000-\u001F\u007F]/gu, ' ')
21
+ .slice(0, 240);
22
+ }
23
+ const COMMAND_CAPTURE_BYTES = 16 * 1024 * 1024;
24
+ // Runs a shell command in the target dir; passed = exit 0. The explicit capture quota
25
+ // avoids Node's 1 MiB default without allowing noisy commands to consume unbounded memory.
26
+ export function commandVerifier(command, options = {}) {
5
27
  return (targetDir) => {
28
+ const phase = options.phase ?? 'verify';
6
29
  try {
7
- execSync(command, { cwd: targetDir, stdio: 'pipe', timeout: 600_000 });
8
- return { passed: true, summary: `verify passed: ${command}` };
30
+ execSync(command, {
31
+ cwd: targetDir,
32
+ stdio: 'pipe',
33
+ timeout: options.timeoutMs ?? 600_000,
34
+ maxBuffer: COMMAND_CAPTURE_BYTES,
35
+ });
36
+ return { passed: true, summary: `${phase} passed: ${command}` };
9
37
  }
10
38
  catch (e) {
11
39
  const err = e;
12
- const out = (err.stderr?.toString('utf8') ?? '') || (err.stdout?.toString('utf8') ?? '');
13
- const tail = out.trim().split('\n').slice(-5).join('\n');
14
- const suffix = err.signal === 'SIGTERM' ? ' (timed out)' : (tail ? `\n${tail}` : '');
15
- return { passed: false, summary: `verify failed: ${command}${suffix}` };
40
+ const captureExceeded = err.code === 'ENOBUFS';
41
+ const captured = labelledOutput(outputText(err.stdout), outputText(err.stderr));
42
+ const captureNotice = `[output truncated: exceeded ${COMMAND_CAPTURE_BYTES}-byte per-stream capture limit]`;
43
+ const raw = captureExceeded
44
+ ? `${captured}${captured ? '\n' : ''}=== capture ===\n${captureNotice}`
45
+ : captured;
46
+ const policy = options.policy ?? DEFAULT_OUTPUT_POLICY;
47
+ const compacted = compactCommandOutput(raw, { previewBytes: policy.previewBytes });
48
+ const timedOut = !captureExceeded && (err.signal === 'SIGTERM' || err.code === 'ETIMEDOUT');
49
+ const qualifier = captureExceeded
50
+ ? ' (capture limit exceeded)'
51
+ : timedOut ? ' (timed out)' : '';
52
+ const parts = [`${phase} failed: ${command}${qualifier}`];
53
+ if (compacted.preview)
54
+ parts.push(compacted.preview);
55
+ if (compacted.originalBytes > policy.artifactThresholdBytes) {
56
+ try {
57
+ const artifact = (options.artifactWriter ?? writeOutputArtifact)(targetDir, raw, {
58
+ phase,
59
+ storyId: process.env.YOKE_STORY,
60
+ });
61
+ parts.push(captureExceeded
62
+ ? artifact.marker.replace('[full output:', '[truncated output:')
63
+ : artifact.marker);
64
+ }
65
+ catch (error) {
66
+ parts.push(`[artifact unavailable: ${errorMessage(error)}]`);
67
+ }
68
+ }
69
+ return { passed: false, summary: parts.join('\n') };
16
70
  }
17
71
  };
18
72
  }
19
73
  /** Execute the proof commands attached to one acceptance criterion. */
20
- export function commandsVerifier(commands) {
74
+ export function commandsVerifier(commands, options = {}) {
21
75
  return (targetDir) => {
22
76
  for (const command of commands) {
23
- const result = commandVerifier(command)(targetDir);
77
+ const result = commandVerifier(command, options)(targetDir);
24
78
  if (!result.passed)
25
79
  return result;
26
80
  }