@hecer/yoke 0.9.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/.codex-plugin/plugin.json +7 -0
  2. package/CHANGELOG.md +169 -149
  3. package/README.md +24 -16
  4. package/TODOS.md +8 -0
  5. package/agents/docs.toml +6 -0
  6. package/agents/implementer.toml +6 -0
  7. package/agents/reviewer.toml +6 -0
  8. package/agents/security.toml +6 -0
  9. package/bench/README.md +45 -42
  10. package/bench/RESULTS.md +46 -36
  11. package/bench/result-schema.mjs +12 -0
  12. package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
  13. package/bench/results/codex-unavailable-1785175418318.json +15 -0
  14. package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
  15. package/bench/run-matrix.mjs +26 -0
  16. package/bench/run.mjs +127 -115
  17. package/canon/loop/prd.schema.md +5 -0
  18. package/canon/manifest.yaml +1 -1
  19. package/canon/skills/authoring-prd/SKILL.md +6 -0
  20. package/canon/skills/ship/SKILL.md +2 -7
  21. package/canon/tools/codex-rtk-hook.mjs +36 -0
  22. package/dist/agents/providers.js +23 -0
  23. package/dist/agents/telemetry.js +30 -0
  24. package/dist/agents/types.js +1 -0
  25. package/dist/audit/changes.js +6 -0
  26. package/dist/audit/command.js +64 -0
  27. package/dist/audit/dependencies.js +21 -0
  28. package/dist/audit/secrets.js +16 -0
  29. package/dist/audit/types.js +1 -0
  30. package/dist/cli.js +22 -4
  31. package/dist/loop/claims.js +57 -0
  32. package/dist/loop/cleanup.js +10 -4
  33. package/dist/loop/git.js +8 -2
  34. package/dist/loop/identity.js +27 -0
  35. package/dist/loop/loop.js +20 -2
  36. package/dist/loop/merge-queue.js +20 -0
  37. package/dist/loop/parallel.js +39 -0
  38. package/dist/loop/prd.js +48 -2
  39. package/dist/loop/run-command.js +51 -6
  40. package/dist/loop/runner.js +41 -29
  41. package/dist/loop/scheduler.js +8 -0
  42. package/dist/prd/command.js +6 -0
  43. package/dist/retrofit/config.js +12 -0
  44. package/dist/retrofit/planners/codex.js +64 -19
  45. package/dist/review/command.js +52 -12
  46. package/dist/review/verdict.js +45 -0
  47. package/docs/MIGRATING-TO-1.0.md +33 -0
  48. package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
  49. package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
  50. package/hooks/hooks.json +19 -0
  51. package/package.json +82 -67
  52. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
  53. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
  54. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
  55. package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
  56. package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
  57. package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
  58. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
  59. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
  60. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
  61. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
  62. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
  63. package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
  64. package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
  65. package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
  66. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
  67. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
  68. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
package/bench/RESULTS.md CHANGED
@@ -1,36 +1,46 @@
1
- # Benchmark results
2
-
3
- Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
4
- [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
5
-
6
- ## Runs
7
-
8
- | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
9
- |---|---|---|---|---|---|---|---|---|
10
- | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
11
- | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
12
- | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
13
-
14
- Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
15
- passed verify on the first iteration; the final quality check (all 16 assertions on the final
16
- tree, outside the loop) is green.
17
-
18
- ## What the harness caught before producing a single number
19
-
20
- Building an honest benchmark is itself a verification pass. The first runs found two real
21
- Yoke bugs, both fixed in 0.3.0:
22
-
23
- 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
24
- 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
25
- 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
26
- `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
27
- stdin (which selects headless mode) and passes only `--yolo`.
28
-
29
- ## Reading the numbers
30
-
31
- - Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
32
- open gap, see the multi-agent design doc).
33
- - N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
34
- compare rows.
35
- - ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
36
- gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
1
+ # Benchmark results
2
+
3
+ Result schema v1 records fixture version, sample label, permission profile, telemetry/model
4
+ availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
5
+
6
+ Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
7
+ [README.md](README.md). One row per run — raw JSON in [`results/`](results/).
8
+
9
+ ## Runs
10
+
11
+ | Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
12
+ |---|---|---|---|---|---|---|---|---|
13
+ | 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
14
+ | 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
15
+ | 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
16
+ | 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
17
+ | 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
18
+ | — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
19
+
20
+ Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
21
+ passed verify on the first iteration; the final quality check (all 16 assertions on the final
22
+ tree, outside the loop) is green.
23
+
24
+ The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
25
+ before changing the fixture, and the Codex executable could not be probed in this Windows
26
+ environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
27
+
28
+ ## What the harness caught before producing a single number
29
+
30
+ Building an honest benchmark is itself a verification pass. The first runs found two real
31
+ Yoke bugs, both fixed in 0.3.0:
32
+
33
+ 1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
34
+ 5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
35
+ 2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
36
+ `-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
37
+ stdin (which selects headless mode) and passes only `--yolo`.
38
+
39
+ ## Reading the numbers
40
+
41
+ - Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
42
+ open gap, see the multi-agent design doc).
43
+ - N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
44
+ compare rows.
45
+ - ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
46
+ gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
@@ -0,0 +1,12 @@
1
+ export const requiredResultFields = [
2
+ 'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
3
+ 'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
4
+ 'iterations', 'finalTestsPass',
5
+ ]
6
+
7
+ export function validateResult(result) {
8
+ const missing = requiredResultFields.filter(key => !(key in result))
9
+ if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
10
+ if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
+ return result
12
+ }
@@ -0,0 +1,50 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "fixtureVersion": "string-kit@1",
4
+ "runner": "claude",
5
+ "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
+ "permissionProfile": "safe",
7
+ "yokeVersion": "1.0.0",
8
+ "fixture": "string-kit",
9
+ "startedAt": "2026-07-27T18:03:26.598Z",
10
+ "wallClockMs": 11345,
11
+ "exitCode": 1,
12
+ "finalState": "blocked",
13
+ "verdict": "blocked",
14
+ "blocker": "Claude runner exited before implementation; fixture verification remained red.",
15
+ "conflicts": 0,
16
+ "iterations": 1,
17
+ "finalTestsPass": false,
18
+ "progress": {
19
+ "passed": 0,
20
+ "total": 3
21
+ },
22
+ "usageAvailable": false,
23
+ "modelAvailable": false,
24
+ "tokens": {
25
+ "inputTokens": 0,
26
+ "outputTokens": 0,
27
+ "model": "<synthetic>"
28
+ },
29
+ "stories": [
30
+ {
31
+ "id": "STORY-1",
32
+ "durationMs": 10954,
33
+ "iterations": 1,
34
+ "finalTestsPass": false
35
+ },
36
+ {
37
+ "id": "STORY-2",
38
+ "durationMs": null,
39
+ "iterations": 0,
40
+ "finalTestsPass": false
41
+ },
42
+ {
43
+ "id": "STORY-3",
44
+ "durationMs": null,
45
+ "iterations": 0,
46
+ "finalTestsPass": false
47
+ }
48
+ ],
49
+ "srcLoc": 3
50
+ }
@@ -0,0 +1,15 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "fixtureVersion": "string-kit@1",
4
+ "runner": "codex",
5
+ "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
+ "permissionProfile": "safe",
7
+ "usageAvailable": false,
8
+ "modelAvailable": false,
9
+ "verdict": "unavailable",
10
+ "blocker": "Zugriff verweigert",
11
+ "conflicts": 0,
12
+ "wallClockMs": null,
13
+ "iterations": 0,
14
+ "finalTestsPass": false
15
+ }
@@ -0,0 +1,46 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "fixtureVersion": "string-kit@1",
4
+ "runner": "gemini",
5
+ "sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
6
+ "permissionProfile": "safe",
7
+ "yokeVersion": "1.0.0",
8
+ "fixture": "string-kit",
9
+ "startedAt": "2026-07-27T18:03:45.223Z",
10
+ "wallClockMs": 13844,
11
+ "exitCode": 1,
12
+ "finalState": "blocked",
13
+ "verdict": "blocked",
14
+ "blocker": "Gemini runner exited before implementation; fixture verification remained red.",
15
+ "conflicts": 0,
16
+ "iterations": 1,
17
+ "finalTestsPass": false,
18
+ "progress": {
19
+ "passed": 0,
20
+ "total": 3
21
+ },
22
+ "usageAvailable": false,
23
+ "modelAvailable": false,
24
+ "tokens": null,
25
+ "stories": [
26
+ {
27
+ "id": "STORY-1",
28
+ "durationMs": 7040,
29
+ "iterations": 1,
30
+ "finalTestsPass": false
31
+ },
32
+ {
33
+ "id": "STORY-2",
34
+ "durationMs": null,
35
+ "iterations": 0,
36
+ "finalTestsPass": false
37
+ },
38
+ {
39
+ "id": "STORY-3",
40
+ "durationMs": null,
41
+ "iterations": 0,
42
+ "finalTestsPass": false
43
+ }
44
+ ],
45
+ "srcLoc": 3
46
+ }
@@ -0,0 +1,26 @@
1
+ #!/usr/bin/env node
2
+ import { spawnSync } from 'node:child_process'
3
+ import { mkdirSync, writeFileSync } from 'node:fs'
4
+ import { dirname, join } from 'node:path'
5
+ import { fileURLToPath } from 'node:url'
6
+ import { validateResult } from './result-schema.mjs'
7
+
8
+ const benchDir = dirname(fileURLToPath(import.meta.url))
9
+ if (process.argv.includes('--help')) {
10
+ console.log('usage: node bench/run-matrix.mjs [--label=sample]')
11
+ process.exit(0)
12
+ }
13
+ const label = process.argv.find(arg => arg.startsWith('--label='))?.slice(8) ?? `matrix-${new Date().toISOString()}`
14
+ mkdirSync(join(benchDir, 'results'), { recursive: true })
15
+ for (const runner of ['claude', 'codex', 'gemini']) {
16
+ const probe = spawnSync(runner, ['--version'], { encoding: 'utf8', shell: process.platform === 'win32', timeout: 20_000 })
17
+ if (probe.status !== 0) {
18
+ const row = validateResult({ schemaVersion: 1, fixtureVersion: 'string-kit@1', runner, sampleLabel: label, permissionProfile: 'safe', usageAvailable: false, modelAvailable: false, verdict: 'unavailable', blocker: (probe.stderr || probe.error?.message || 'CLI unavailable').trim(), conflicts: 0, wallClockMs: null, iterations: 0, finalTestsPass: false })
19
+ const out = join(benchDir, 'results', `${runner}-unavailable-${Date.now()}.json`)
20
+ writeFileSync(out, JSON.stringify(row, null, 2) + '\n')
21
+ console.log(JSON.stringify(row))
22
+ continue
23
+ }
24
+ const run = spawnSync(process.execPath, [join(benchDir, 'run.mjs'), `--runner=${runner}`, `--label=${label}`], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'inherit'] })
25
+ if (run.stdout) process.stdout.write(run.stdout)
26
+ }
package/bench/run.mjs CHANGED
@@ -1,115 +1,127 @@
1
- #!/usr/bin/env node
2
- // Yoke benchmark harness.
3
- //
4
- // node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
5
- //
6
- // Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
7
- // `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
8
- // durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
9
- // the loop's token hook (claude runner only), and quality as the fixture's own
10
- // pre-written tests — run per story AFTER the loop finishes, on the final tree.
11
- import { spawn, spawnSync } from 'node:child_process'
12
- import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
13
- import { join, dirname } from 'node:path'
14
- import { fileURLToPath } from 'node:url'
15
-
16
- const benchDir = dirname(fileURLToPath(import.meta.url))
17
- const repoRoot = dirname(benchDir)
18
- const cli = join(repoRoot, 'dist', 'cli.js')
19
-
20
- const args = Object.fromEntries(
21
- process.argv.slice(2).filter(a => a.startsWith('--')).map(a => {
22
- const [k, v] = a.slice(2).split('=')
23
- return [k, v ?? true]
24
- }),
25
- )
26
- const runner = args.runner
27
- if (!['claude', 'codex', 'gemini'].includes(runner)) {
28
- console.error('usage: node bench/run.mjs --runner=<claude|codex|gemini> [--max=6] [--timeout=10] [--label=note]')
29
- process.exit(2)
30
- }
31
- const max = Number(args.max ?? 6)
32
- const timeout = Number(args.timeout ?? 10)
33
-
34
- const stamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
35
- const runDir = join(benchDir, '.runs', `${runner}-${stamp}`)
36
- mkdirSync(runDir, { recursive: true })
37
- cpSync(join(benchDir, 'fixtures', 'string-kit'), runDir, { recursive: true })
38
-
39
- const git = (...a) => {
40
- const r = spawnSync('git', ['-C', runDir, ...a], { encoding: 'utf8' })
41
- if (r.status !== 0) throw new Error(`git ${a.join(' ')} failed: ${r.stderr}`)
42
- }
43
- git('init', '-q')
44
- git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'add', '-A')
45
- git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'commit', '-q', '-m', 'bench: fixture baseline')
46
-
47
- // A nested Claude Code session refuses some operations; scrub session markers.
48
- const env = { ...process.env }
49
- for (const k of Object.keys(env)) if (k.startsWith('CLAUDE_CODE') || k === 'CLAUDECODE') delete env[k]
50
-
51
- console.error(`[bench] ${runner} → ${runDir}`)
52
- const t0 = Date.now()
53
- const events = []
54
- const child = spawn(process.execPath, [cli, 'loop', 'run', runDir, '--json', `--runner=${runner}`, `--max=${max}`, `--timeout=${timeout}`], {
55
- env, stdio: ['ignore', 'pipe', 'inherit'],
56
- })
57
- let buf = ''
58
- child.stdout.on('data', d => {
59
- buf += d
60
- let i
61
- while ((i = buf.indexOf('\n')) >= 0) {
62
- const line = buf.slice(0, i).trim()
63
- buf = buf.slice(i + 1)
64
- if (!line) continue
65
- try { events.push({ at: Date.now(), ...JSON.parse(line) }) } catch { /* non-JSON noise */ }
66
- }
67
- })
68
- const exitCode = await new Promise(res => child.on('close', res))
69
- const wallClockMs = Date.now() - t0
70
-
71
- // Per-story duration: first event mentioning the story -> first event mentioning the next story (or end).
72
- const storyIds = ['STORY-1', 'STORY-2', 'STORY-3']
73
- const firstSeen = {}
74
- for (const e of events) if (e.story && !(e.story in firstSeen)) firstSeen[e.story] = e.at
75
- const stories = storyIds.map((id, idx) => {
76
- const start = firstSeen[id]
77
- const next = storyIds.slice(idx + 1).map(n => firstSeen[n]).find(v => v !== undefined)
78
- const durationMs = start === undefined ? null : (next ?? t0 + wallClockMs) - start
79
- const iterations = new Set(events.filter(e => e.story === id).map(e => e.iteration)).size
80
- // Quality: the fixture's own tests for this story, on the final tree.
81
- const q = spawnSync(process.execPath, ['--test', `tests/${id}.test.mjs`], { cwd: runDir, encoding: 'utf8' })
82
- return { id, durationMs, iterations, finalTestsPass: q.status === 0 }
83
- })
84
-
85
- const last = events[events.length - 1] ?? {}
86
- let status = {}
87
- try { status = JSON.parse(readFileSync(join(runDir, '.yoke', 'loop-status.json'), 'utf8')) } catch { /* loop may have refused before writing status */ }
88
-
89
- // Source size (LOC in src/) as a code-economy proxy.
90
- const loc = (dir) => readdirSync(dir).reduce((n, f) => {
91
- const p = join(dir, f)
92
- if (statSync(p).isDirectory()) return n + loc(p)
93
- return n + readFileSync(p, 'utf8').split('\n').filter(l => l.trim() !== '').length
94
- }, 0)
95
-
96
- const result = {
97
- runner,
98
- label: args.label ?? null,
99
- yokeVersion: JSON.parse(readFileSync(join(repoRoot, 'package.json'), 'utf8')).version,
100
- fixture: 'string-kit',
101
- startedAt: new Date(t0).toISOString(),
102
- wallClockMs,
103
- exitCode,
104
- finalState: last.state ?? null,
105
- progress: last.progress ?? null,
106
- tokens: status.tokens ?? null, // claude runner only; gemini/codex report none (documented gap)
107
- stories,
108
- srcLoc: loc(join(runDir, 'src')),
109
- }
110
-
111
- mkdirSync(join(benchDir, 'results'), { recursive: true })
112
- const out = join(benchDir, 'results', `${runner}-${stamp}.json`)
113
- writeFileSync(out, JSON.stringify(result, null, 2) + '\n')
114
- console.error(`[bench] done: ${out}`)
115
- console.log(JSON.stringify(result, null, 2))
1
+ #!/usr/bin/env node
2
+ // Yoke benchmark harness.
3
+ //
4
+ // node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
5
+ //
6
+ // Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
7
+ // `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
8
+ // durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
9
+ // the loop's token hook (claude runner only), and quality as the fixture's own
10
+ // pre-written tests — run per story AFTER the loop finishes, on the final tree.
11
+ import { spawn, spawnSync } from 'node:child_process'
12
+ import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
13
+ import { join, dirname } from 'node:path'
14
+ import { fileURLToPath } from 'node:url'
15
+ import { validateResult } from './result-schema.mjs'
16
+
17
+ const benchDir = dirname(fileURLToPath(import.meta.url))
18
+ const repoRoot = dirname(benchDir)
19
+ const cli = join(repoRoot, 'dist', 'cli.js')
20
+
21
+ const args = Object.fromEntries(
22
+ process.argv.slice(2).filter(a => a.startsWith('--')).map(a => {
23
+ const [k, v] = a.slice(2).split('=')
24
+ return [k, v ?? true]
25
+ }),
26
+ )
27
+ const runner = args.runner
28
+ if (!['claude', 'codex', 'gemini'].includes(runner)) {
29
+ console.error('usage: node bench/run.mjs --runner=<claude|codex|gemini> [--max=6] [--timeout=10] [--label=note]')
30
+ process.exit(2)
31
+ }
32
+ const max = Number(args.max ?? 6)
33
+ const timeout = Number(args.timeout ?? 10)
34
+
35
+ const stamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
36
+ const runDir = join(benchDir, '.runs', `${runner}-${stamp}`)
37
+ mkdirSync(runDir, { recursive: true })
38
+ cpSync(join(benchDir, 'fixtures', 'string-kit'), runDir, { recursive: true })
39
+
40
+ const git = (...a) => {
41
+ const r = spawnSync('git', ['-C', runDir, ...a], { encoding: 'utf8' })
42
+ if (r.status !== 0) throw new Error(`git ${a.join(' ')} failed: ${r.stderr}`)
43
+ }
44
+ git('init', '-q')
45
+ git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'add', '-A')
46
+ git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'commit', '-q', '-m', 'bench: fixture baseline')
47
+
48
+ // A nested Claude Code session refuses some operations; scrub session markers.
49
+ const env = { ...process.env }
50
+ for (const k of Object.keys(env)) if (k.startsWith('CLAUDE_CODE') || k === 'CLAUDECODE') delete env[k]
51
+
52
+ console.error(`[bench] ${runner} → ${runDir}`)
53
+ const t0 = Date.now()
54
+ const events = []
55
+ const child = spawn(process.execPath, [cli, 'loop', 'run', runDir, '--json', `--runner=${runner}`, `--max=${max}`, `--timeout=${timeout}`], {
56
+ env, stdio: ['ignore', 'pipe', 'inherit'],
57
+ })
58
+ let buf = ''
59
+ child.stdout.on('data', d => {
60
+ buf += d
61
+ let i
62
+ while ((i = buf.indexOf('\n')) >= 0) {
63
+ const line = buf.slice(0, i).trim()
64
+ buf = buf.slice(i + 1)
65
+ if (!line) continue
66
+ try { events.push({ at: Date.now(), ...JSON.parse(line) }) } catch { /* non-JSON noise */ }
67
+ }
68
+ })
69
+ const exitCode = await new Promise(res => child.on('close', res))
70
+ const wallClockMs = Date.now() - t0
71
+
72
+ // Per-story duration: first event mentioning the story -> first event mentioning the next story (or end).
73
+ const storyIds = ['STORY-1', 'STORY-2', 'STORY-3']
74
+ const firstSeen = {}
75
+ for (const e of events) if (e.story && !(e.story in firstSeen)) firstSeen[e.story] = e.at
76
+ const stories = storyIds.map((id, idx) => {
77
+ const start = firstSeen[id]
78
+ const next = storyIds.slice(idx + 1).map(n => firstSeen[n]).find(v => v !== undefined)
79
+ const durationMs = start === undefined ? null : (next ?? t0 + wallClockMs) - start
80
+ const iterations = new Set(events.filter(e => e.story === id).map(e => e.iteration)).size
81
+ // Quality: the fixture's own tests for this story, on the final tree.
82
+ const q = spawnSync(process.execPath, ['--test', `tests/${id}.test.mjs`], { cwd: runDir, encoding: 'utf8' })
83
+ return { id, durationMs, iterations, finalTestsPass: q.status === 0 }
84
+ })
85
+
86
+ const last = events[events.length - 1] ?? {}
87
+ let status = {}
88
+ try { status = JSON.parse(readFileSync(join(runDir, '.yoke', 'loop-status.json'), 'utf8')) } catch { /* loop may have refused before writing status */ }
89
+
90
+ // Source size (LOC in src/) as a code-economy proxy.
91
+ const loc = (dir) => readdirSync(dir).reduce((n, f) => {
92
+ const p = join(dir, f)
93
+ if (statSync(p).isDirectory()) return n + loc(p)
94
+ return n + readFileSync(p, 'utf8').split('\n').filter(l => l.trim() !== '').length
95
+ }, 0)
96
+
97
+ const result = {
98
+ schemaVersion: 1,
99
+ fixtureVersion: 'string-kit@1',
100
+ runner,
101
+ sampleLabel: String(args.label ?? `${runner}-${stamp}`),
102
+ permissionProfile: args.unsafe ? 'unsafe' : 'safe',
103
+ yokeVersion: JSON.parse(readFileSync(join(repoRoot, 'package.json'), 'utf8')).version,
104
+ fixture: 'string-kit',
105
+ startedAt: new Date(t0).toISOString(),
106
+ wallClockMs,
107
+ exitCode,
108
+ finalState: last.state ?? null,
109
+ verdict: exitCode === 0 ? 'completed' : (/api key|login|auth/i.test(String(status.reason ?? '')) ? 'auth-failed' : 'blocked'),
110
+ blocker: exitCode === 0 ? null : (status.reason ?? 'runner exited without a diagnostic'),
111
+ conflicts: events.filter(e => /conflict/i.test(String(e.reason ?? e.summary ?? ''))).length,
112
+ iterations: stories.reduce((sum, story) => sum + story.iterations, 0),
113
+ finalTestsPass: stories.every(story => story.finalTestsPass),
114
+ progress: last.progress ?? null,
115
+ usageAvailable: Number(status.tokens?.inputTokens ?? 0) + Number(status.tokens?.outputTokens ?? 0) > 0,
116
+ modelAvailable: typeof status.tokens?.model === 'string' && status.tokens.model !== '<synthetic>',
117
+ tokens: status.tokens ?? null,
118
+ stories,
119
+ srcLoc: loc(join(runDir, 'src')),
120
+ }
121
+ validateResult(result)
122
+
123
+ mkdirSync(join(benchDir, 'results'), { recursive: true })
124
+ const out = join(benchDir, 'results', `${runner}-${stamp}.json`)
125
+ writeFileSync(out, JSON.stringify(result, null, 2) + '\n')
126
+ console.error(`[bench] done: ${out}`)
127
+ console.log(JSON.stringify(result, null, 2))
@@ -6,9 +6,14 @@ The loop is driven by a versioned PRD file. Each story:
6
6
  - id: STORY-1
7
7
  title: Short imperative description
8
8
  priority: 1 # lower = higher priority
9
+ needs: [] # optional dependency IDs; no unknown IDs, self-links, or cycles
10
+ area: api # optional collision domain for parallel scheduling
11
+ agent: codex # optional claude|codex|gemini affinity
9
12
  acceptance: # Definition of Done (required before implementation)
10
13
  - The endpoint returns 200 for a valid request.
11
14
  passes: false # set true only when acceptance is met and tests are green
12
15
  ```
13
16
 
14
17
  Stop condition: every story has `passes: true`.
18
+
19
+ Stories without `needs`, `area`, or `agent` retain the serial pre-1.0 behavior. A story is ready only when every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously active areas, and uses `agent` as an affinity hint.
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 0.1.0
2
+ version: 1.0.0
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology }
@@ -29,6 +29,9 @@ good stories (small, testable, ordered) let it run overnight.
29
29
  has nobody to ask. A criterion that still needs a decision ("TBD", "choose a provider")
30
30
  is not loop-ready; resolve it here or the agent will either guess (default) or block
31
31
  (`--on-ambiguity=abort`).
32
+ 8. **Model real dependencies.** Add `needs` only for hard prerequisites, `area` for files or
33
+ subsystems that must not be edited concurrently, and `agent` only as an affinity hint.
34
+ Dependency IDs must exist; self-dependencies and cycles are invalid.
32
35
 
33
36
  ## Format (`.yoke/prd.yaml`)
34
37
 
@@ -43,6 +46,9 @@ good stories (small, testable, ordered) let it run overnight.
43
46
  - id: STORY-2
44
47
  title: add the sum command
45
48
  priority: 2
49
+ needs: [STORY-1]
50
+ area: cli
51
+ agent: codex
46
52
  acceptance:
47
53
  - "cli sum 1 2 prints 3"
48
54
  - "non-numeric input exits 1 with an error message"
@@ -526,15 +526,10 @@ Analyze the diff and group changes into logical commits. Each commit should repr
526
526
 
527
527
  **Each commit must be independently valid** — no broken imports, no references to code that doesn't exist yet.
528
528
 
529
- The **final commit** (VERSION + CHANGELOG) gets the version tag and co-author trailer:
529
+ The **final commit** contains VERSION + CHANGELOG. The project's commit identity and co-author policy always wins; never add an AI co-author trailer unless the project explicitly allows it.
530
530
 
531
531
  ```bash
532
- git commit -m "$(cat <<'EOF'
533
- chore: bump version and changelog (vX.Y.Z.W)
534
-
535
- Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
536
- EOF
537
- )"
532
+ git commit -m "chore: bump version and changelog (vX.Y.Z.W)"
538
533
  ```
539
534
 
540
535
  ---
@@ -0,0 +1,36 @@
1
+ import { spawnSync } from 'node:child_process'
2
+ import { resolve } from 'node:path'
3
+ import { pathToFileURL } from 'node:url'
4
+
5
+ function rtkCheck(command) {
6
+ const result = spawnSync('rtk', ['hook', 'check', command], { encoding: 'utf8', timeout: 3000 })
7
+ return result.status === 0 ? result.stdout.trim() : ''
8
+ }
9
+
10
+ export function rewriteHookInput(input, check = rtkCheck) {
11
+ if (input?.tool_name !== 'Bash' && input?.toolName !== 'Bash') return null
12
+ const toolInput = input.tool_input ?? input.toolInput
13
+ const command = toolInput?.command
14
+ if (typeof command !== 'string' || command.trim() === '') return null
15
+ const rewritten = check(command)
16
+ if (!rewritten || rewritten === command) return null
17
+ return {
18
+ hookSpecificOutput: {
19
+ hookEventName: 'PreToolUse',
20
+ updatedInput: { ...toolInput, command: rewritten },
21
+ },
22
+ }
23
+ }
24
+
25
+ async function main() {
26
+ let raw = ''
27
+ for await (const chunk of process.stdin) raw += chunk
28
+ try {
29
+ const output = rewriteHookInput(JSON.parse(raw))
30
+ if (output) process.stdout.write(JSON.stringify(output))
31
+ } catch {
32
+ // Compression is an optimization. Malformed input must never block Codex.
33
+ }
34
+ }
35
+
36
+ if (process.argv[1] && pathToFileURL(resolve(process.argv[1])).href === import.meta.url) await main()
@@ -0,0 +1,23 @@
1
+ const argsFor = (agent, permissions) => {
2
+ if (agent === 'claude') {
3
+ const mode = permissions === 'unsafe' ? 'bypassPermissions' : permissions === 'read-only' ? 'plan' : 'auto';
4
+ const args = ['-p', '--permission-mode', mode];
5
+ if (permissions === 'unsafe')
6
+ args.push('--dangerously-skip-permissions');
7
+ return [...args, '--output-format', 'stream-json', '--verbose'];
8
+ }
9
+ if (agent === 'codex') {
10
+ if (permissions === 'unsafe')
11
+ return ['exec', '--dangerously-bypass-approvals-and-sandbox', '--json'];
12
+ if (permissions === 'read-only')
13
+ return ['exec', '--sandbox', 'read-only', '--json'];
14
+ return ['exec', '--full-auto', '--json'];
15
+ }
16
+ if (permissions === 'unsafe')
17
+ return ['--yolo', '--output-format', 'stream-json'];
18
+ const approval = permissions === 'read-only' ? 'plan' : 'auto_edit';
19
+ return ['--approval-mode', approval, '--sandbox', '--output-format', 'stream-json'];
20
+ };
21
+ export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe') {
22
+ return { command: agent, args: argsFor(agent, permissions), input: prompt, cwd };
23
+ }