@hecer/yoke 0.9.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +18 -0
- package/.claude-plugin/plugin.json +13 -0
- package/.codex-plugin/plugin.json +7 -0
- package/CHANGELOG.md +192 -149
- package/README.md +101 -51
- package/TODOS.md +8 -0
- package/agents/docs.toml +6 -0
- package/agents/implementer.toml +6 -0
- package/agents/reviewer.toml +6 -0
- package/agents/security.toml +6 -0
- package/bench/README.md +45 -42
- package/bench/RESULTS.md +46 -36
- package/bench/result-schema.mjs +12 -0
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
- package/bench/results/codex-unavailable-1785175418318.json +15 -0
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
- package/bench/run-matrix.mjs +26 -0
- package/bench/run.mjs +127 -115
- package/canon/AGENTS.md +2 -0
- package/canon/loop/loop-spec.md +4 -2
- package/canon/loop/prd.schema.md +5 -0
- package/canon/manifest.yaml +2 -1
- package/canon/skills/authoring-prd/SKILL.md +10 -3
- package/canon/skills/ship/SKILL.md +2 -7
- package/canon/skills/workflow/SKILL.md +4 -0
- package/canon/skills/yoke-retrofit/SKILL.md +18 -11
- package/canon/skills/yoke-workflow/SKILL.md +20 -0
- package/canon/tools/codex-rtk-hook.mjs +36 -0
- package/dist/agents/host.js +26 -0
- package/dist/agents/providers.js +23 -0
- package/dist/agents/telemetry.js +30 -0
- package/dist/agents/types.js +1 -0
- package/dist/audit/changes.js +6 -0
- package/dist/audit/command.js +64 -0
- package/dist/audit/dependencies.js +21 -0
- package/dist/audit/secrets.js +16 -0
- package/dist/audit/types.js +1 -0
- package/dist/cli.js +189 -6
- package/dist/context/context.js +15 -2
- package/dist/loop/claims.js +57 -0
- package/dist/loop/cleanup.js +98 -27
- package/dist/loop/decision.js +517 -0
- package/dist/loop/git.js +31 -2
- package/dist/loop/identity.js +27 -0
- package/dist/loop/lock.js +104 -13
- package/dist/loop/loop.js +49 -2
- package/dist/loop/merge-queue.js +20 -0
- package/dist/loop/parallel.js +39 -0
- package/dist/loop/prd.js +48 -2
- package/dist/loop/run-command.js +118 -12
- package/dist/loop/runner.js +48 -30
- package/dist/loop/scheduler.js +8 -0
- package/dist/prd/command.js +30 -21
- package/dist/retrofit/command.js +3 -2
- package/dist/retrofit/config.js +16 -0
- package/dist/retrofit/gitignore.js +8 -0
- package/dist/retrofit/planners/codex.js +64 -19
- package/dist/retrofit/report.js +1 -1
- package/dist/review/command.js +52 -12
- package/dist/review/verdict.js +45 -0
- package/dist/setup/command.js +82 -0
- package/docs/MIGRATING-TO-1.0.md +33 -0
- package/docs/MIGRATING-TO-1.1.md +27 -0
- package/docs/PUBLISHING.md +77 -41
- package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
- package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
- package/gemini-extension.json +6 -0
- package/hooks/hooks.json +19 -0
- package/package.json +84 -67
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
- package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
- package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
- package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
- package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
- package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "claude",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"yokeVersion": "1.0.0",
|
|
8
|
+
"fixture": "string-kit",
|
|
9
|
+
"startedAt": "2026-07-27T18:03:26.598Z",
|
|
10
|
+
"wallClockMs": 11345,
|
|
11
|
+
"exitCode": 1,
|
|
12
|
+
"finalState": "blocked",
|
|
13
|
+
"verdict": "blocked",
|
|
14
|
+
"blocker": "Claude runner exited before implementation; fixture verification remained red.",
|
|
15
|
+
"conflicts": 0,
|
|
16
|
+
"iterations": 1,
|
|
17
|
+
"finalTestsPass": false,
|
|
18
|
+
"progress": {
|
|
19
|
+
"passed": 0,
|
|
20
|
+
"total": 3
|
|
21
|
+
},
|
|
22
|
+
"usageAvailable": false,
|
|
23
|
+
"modelAvailable": false,
|
|
24
|
+
"tokens": {
|
|
25
|
+
"inputTokens": 0,
|
|
26
|
+
"outputTokens": 0,
|
|
27
|
+
"model": "<synthetic>"
|
|
28
|
+
},
|
|
29
|
+
"stories": [
|
|
30
|
+
{
|
|
31
|
+
"id": "STORY-1",
|
|
32
|
+
"durationMs": 10954,
|
|
33
|
+
"iterations": 1,
|
|
34
|
+
"finalTestsPass": false
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"id": "STORY-2",
|
|
38
|
+
"durationMs": null,
|
|
39
|
+
"iterations": 0,
|
|
40
|
+
"finalTestsPass": false
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"id": "STORY-3",
|
|
44
|
+
"durationMs": null,
|
|
45
|
+
"iterations": 0,
|
|
46
|
+
"finalTestsPass": false
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"srcLoc": 3
|
|
50
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "codex",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"usageAvailable": false,
|
|
8
|
+
"modelAvailable": false,
|
|
9
|
+
"verdict": "unavailable",
|
|
10
|
+
"blocker": "Zugriff verweigert",
|
|
11
|
+
"conflicts": 0,
|
|
12
|
+
"wallClockMs": null,
|
|
13
|
+
"iterations": 0,
|
|
14
|
+
"finalTestsPass": false
|
|
15
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "gemini",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"yokeVersion": "1.0.0",
|
|
8
|
+
"fixture": "string-kit",
|
|
9
|
+
"startedAt": "2026-07-27T18:03:45.223Z",
|
|
10
|
+
"wallClockMs": 13844,
|
|
11
|
+
"exitCode": 1,
|
|
12
|
+
"finalState": "blocked",
|
|
13
|
+
"verdict": "blocked",
|
|
14
|
+
"blocker": "Gemini runner exited before implementation; fixture verification remained red.",
|
|
15
|
+
"conflicts": 0,
|
|
16
|
+
"iterations": 1,
|
|
17
|
+
"finalTestsPass": false,
|
|
18
|
+
"progress": {
|
|
19
|
+
"passed": 0,
|
|
20
|
+
"total": 3
|
|
21
|
+
},
|
|
22
|
+
"usageAvailable": false,
|
|
23
|
+
"modelAvailable": false,
|
|
24
|
+
"tokens": null,
|
|
25
|
+
"stories": [
|
|
26
|
+
{
|
|
27
|
+
"id": "STORY-1",
|
|
28
|
+
"durationMs": 7040,
|
|
29
|
+
"iterations": 1,
|
|
30
|
+
"finalTestsPass": false
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"id": "STORY-2",
|
|
34
|
+
"durationMs": null,
|
|
35
|
+
"iterations": 0,
|
|
36
|
+
"finalTestsPass": false
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"id": "STORY-3",
|
|
40
|
+
"durationMs": null,
|
|
41
|
+
"iterations": 0,
|
|
42
|
+
"finalTestsPass": false
|
|
43
|
+
}
|
|
44
|
+
],
|
|
45
|
+
"srcLoc": 3
|
|
46
|
+
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { spawnSync } from 'node:child_process'
|
|
3
|
+
import { mkdirSync, writeFileSync } from 'node:fs'
|
|
4
|
+
import { dirname, join } from 'node:path'
|
|
5
|
+
import { fileURLToPath } from 'node:url'
|
|
6
|
+
import { validateResult } from './result-schema.mjs'
|
|
7
|
+
|
|
8
|
+
const benchDir = dirname(fileURLToPath(import.meta.url))
|
|
9
|
+
if (process.argv.includes('--help')) {
|
|
10
|
+
console.log('usage: node bench/run-matrix.mjs [--label=sample]')
|
|
11
|
+
process.exit(0)
|
|
12
|
+
}
|
|
13
|
+
const label = process.argv.find(arg => arg.startsWith('--label='))?.slice(8) ?? `matrix-${new Date().toISOString()}`
|
|
14
|
+
mkdirSync(join(benchDir, 'results'), { recursive: true })
|
|
15
|
+
for (const runner of ['claude', 'codex', 'gemini']) {
|
|
16
|
+
const probe = spawnSync(runner, ['--version'], { encoding: 'utf8', shell: process.platform === 'win32', timeout: 20_000 })
|
|
17
|
+
if (probe.status !== 0) {
|
|
18
|
+
const row = validateResult({ schemaVersion: 1, fixtureVersion: 'string-kit@1', runner, sampleLabel: label, permissionProfile: 'safe', usageAvailable: false, modelAvailable: false, verdict: 'unavailable', blocker: (probe.stderr || probe.error?.message || 'CLI unavailable').trim(), conflicts: 0, wallClockMs: null, iterations: 0, finalTestsPass: false })
|
|
19
|
+
const out = join(benchDir, 'results', `${runner}-unavailable-${Date.now()}.json`)
|
|
20
|
+
writeFileSync(out, JSON.stringify(row, null, 2) + '\n')
|
|
21
|
+
console.log(JSON.stringify(row))
|
|
22
|
+
continue
|
|
23
|
+
}
|
|
24
|
+
const run = spawnSync(process.execPath, [join(benchDir, 'run.mjs'), `--runner=${runner}`, `--label=${label}`], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'inherit'] })
|
|
25
|
+
if (run.stdout) process.stdout.write(run.stdout)
|
|
26
|
+
}
|
package/bench/run.mjs
CHANGED
|
@@ -1,115 +1,127 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
// Yoke benchmark harness.
|
|
3
|
-
//
|
|
4
|
-
// node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
|
|
5
|
-
//
|
|
6
|
-
// Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
|
|
7
|
-
// `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
|
|
8
|
-
// durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
|
|
9
|
-
// the loop's token hook (claude runner only), and quality as the fixture's own
|
|
10
|
-
// pre-written tests — run per story AFTER the loop finishes, on the final tree.
|
|
11
|
-
import { spawn, spawnSync } from 'node:child_process'
|
|
12
|
-
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
|
|
13
|
-
import { join, dirname } from 'node:path'
|
|
14
|
-
import { fileURLToPath } from 'node:url'
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
const
|
|
18
|
-
const
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
)
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
const
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
const
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
git('
|
|
45
|
-
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', '
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
const
|
|
54
|
-
const
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
const
|
|
74
|
-
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
const
|
|
78
|
-
const
|
|
79
|
-
const
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Yoke benchmark harness.
|
|
3
|
+
//
|
|
4
|
+
// node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
|
|
5
|
+
//
|
|
6
|
+
// Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
|
|
7
|
+
// `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
|
|
8
|
+
// durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
|
|
9
|
+
// the loop's token hook (claude runner only), and quality as the fixture's own
|
|
10
|
+
// pre-written tests — run per story AFTER the loop finishes, on the final tree.
|
|
11
|
+
import { spawn, spawnSync } from 'node:child_process'
|
|
12
|
+
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
|
|
13
|
+
import { join, dirname } from 'node:path'
|
|
14
|
+
import { fileURLToPath } from 'node:url'
|
|
15
|
+
import { validateResult } from './result-schema.mjs'
|
|
16
|
+
|
|
17
|
+
const benchDir = dirname(fileURLToPath(import.meta.url))
|
|
18
|
+
const repoRoot = dirname(benchDir)
|
|
19
|
+
const cli = join(repoRoot, 'dist', 'cli.js')
|
|
20
|
+
|
|
21
|
+
const args = Object.fromEntries(
|
|
22
|
+
process.argv.slice(2).filter(a => a.startsWith('--')).map(a => {
|
|
23
|
+
const [k, v] = a.slice(2).split('=')
|
|
24
|
+
return [k, v ?? true]
|
|
25
|
+
}),
|
|
26
|
+
)
|
|
27
|
+
const runner = args.runner
|
|
28
|
+
if (!['claude', 'codex', 'gemini'].includes(runner)) {
|
|
29
|
+
console.error('usage: node bench/run.mjs --runner=<claude|codex|gemini> [--max=6] [--timeout=10] [--label=note]')
|
|
30
|
+
process.exit(2)
|
|
31
|
+
}
|
|
32
|
+
const max = Number(args.max ?? 6)
|
|
33
|
+
const timeout = Number(args.timeout ?? 10)
|
|
34
|
+
|
|
35
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
|
|
36
|
+
const runDir = join(benchDir, '.runs', `${runner}-${stamp}`)
|
|
37
|
+
mkdirSync(runDir, { recursive: true })
|
|
38
|
+
cpSync(join(benchDir, 'fixtures', 'string-kit'), runDir, { recursive: true })
|
|
39
|
+
|
|
40
|
+
const git = (...a) => {
|
|
41
|
+
const r = spawnSync('git', ['-C', runDir, ...a], { encoding: 'utf8' })
|
|
42
|
+
if (r.status !== 0) throw new Error(`git ${a.join(' ')} failed: ${r.stderr}`)
|
|
43
|
+
}
|
|
44
|
+
git('init', '-q')
|
|
45
|
+
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'add', '-A')
|
|
46
|
+
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'commit', '-q', '-m', 'bench: fixture baseline')
|
|
47
|
+
|
|
48
|
+
// A nested Claude Code session refuses some operations; scrub session markers.
|
|
49
|
+
const env = { ...process.env }
|
|
50
|
+
for (const k of Object.keys(env)) if (k.startsWith('CLAUDE_CODE') || k === 'CLAUDECODE') delete env[k]
|
|
51
|
+
|
|
52
|
+
console.error(`[bench] ${runner} → ${runDir}`)
|
|
53
|
+
const t0 = Date.now()
|
|
54
|
+
const events = []
|
|
55
|
+
const child = spawn(process.execPath, [cli, 'loop', 'run', runDir, '--json', `--runner=${runner}`, `--max=${max}`, `--timeout=${timeout}`], {
|
|
56
|
+
env, stdio: ['ignore', 'pipe', 'inherit'],
|
|
57
|
+
})
|
|
58
|
+
let buf = ''
|
|
59
|
+
child.stdout.on('data', d => {
|
|
60
|
+
buf += d
|
|
61
|
+
let i
|
|
62
|
+
while ((i = buf.indexOf('\n')) >= 0) {
|
|
63
|
+
const line = buf.slice(0, i).trim()
|
|
64
|
+
buf = buf.slice(i + 1)
|
|
65
|
+
if (!line) continue
|
|
66
|
+
try { events.push({ at: Date.now(), ...JSON.parse(line) }) } catch { /* non-JSON noise */ }
|
|
67
|
+
}
|
|
68
|
+
})
|
|
69
|
+
const exitCode = await new Promise(res => child.on('close', res))
|
|
70
|
+
const wallClockMs = Date.now() - t0
|
|
71
|
+
|
|
72
|
+
// Per-story duration: first event mentioning the story -> first event mentioning the next story (or end).
|
|
73
|
+
const storyIds = ['STORY-1', 'STORY-2', 'STORY-3']
|
|
74
|
+
const firstSeen = {}
|
|
75
|
+
for (const e of events) if (e.story && !(e.story in firstSeen)) firstSeen[e.story] = e.at
|
|
76
|
+
const stories = storyIds.map((id, idx) => {
|
|
77
|
+
const start = firstSeen[id]
|
|
78
|
+
const next = storyIds.slice(idx + 1).map(n => firstSeen[n]).find(v => v !== undefined)
|
|
79
|
+
const durationMs = start === undefined ? null : (next ?? t0 + wallClockMs) - start
|
|
80
|
+
const iterations = new Set(events.filter(e => e.story === id).map(e => e.iteration)).size
|
|
81
|
+
// Quality: the fixture's own tests for this story, on the final tree.
|
|
82
|
+
const q = spawnSync(process.execPath, ['--test', `tests/${id}.test.mjs`], { cwd: runDir, encoding: 'utf8' })
|
|
83
|
+
return { id, durationMs, iterations, finalTestsPass: q.status === 0 }
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
const last = events[events.length - 1] ?? {}
|
|
87
|
+
let status = {}
|
|
88
|
+
try { status = JSON.parse(readFileSync(join(runDir, '.yoke', 'loop-status.json'), 'utf8')) } catch { /* loop may have refused before writing status */ }
|
|
89
|
+
|
|
90
|
+
// Source size (LOC in src/) as a code-economy proxy.
|
|
91
|
+
const loc = (dir) => readdirSync(dir).reduce((n, f) => {
|
|
92
|
+
const p = join(dir, f)
|
|
93
|
+
if (statSync(p).isDirectory()) return n + loc(p)
|
|
94
|
+
return n + readFileSync(p, 'utf8').split('\n').filter(l => l.trim() !== '').length
|
|
95
|
+
}, 0)
|
|
96
|
+
|
|
97
|
+
const result = {
|
|
98
|
+
schemaVersion: 1,
|
|
99
|
+
fixtureVersion: 'string-kit@1',
|
|
100
|
+
runner,
|
|
101
|
+
sampleLabel: String(args.label ?? `${runner}-${stamp}`),
|
|
102
|
+
permissionProfile: args.unsafe ? 'unsafe' : 'safe',
|
|
103
|
+
yokeVersion: JSON.parse(readFileSync(join(repoRoot, 'package.json'), 'utf8')).version,
|
|
104
|
+
fixture: 'string-kit',
|
|
105
|
+
startedAt: new Date(t0).toISOString(),
|
|
106
|
+
wallClockMs,
|
|
107
|
+
exitCode,
|
|
108
|
+
finalState: last.state ?? null,
|
|
109
|
+
verdict: exitCode === 0 ? 'completed' : (/api key|login|auth/i.test(String(status.reason ?? '')) ? 'auth-failed' : 'blocked'),
|
|
110
|
+
blocker: exitCode === 0 ? null : (status.reason ?? 'runner exited without a diagnostic'),
|
|
111
|
+
conflicts: events.filter(e => /conflict/i.test(String(e.reason ?? e.summary ?? ''))).length,
|
|
112
|
+
iterations: stories.reduce((sum, story) => sum + story.iterations, 0),
|
|
113
|
+
finalTestsPass: stories.every(story => story.finalTestsPass),
|
|
114
|
+
progress: last.progress ?? null,
|
|
115
|
+
usageAvailable: Number(status.tokens?.inputTokens ?? 0) + Number(status.tokens?.outputTokens ?? 0) > 0,
|
|
116
|
+
modelAvailable: typeof status.tokens?.model === 'string' && status.tokens.model !== '<synthetic>',
|
|
117
|
+
tokens: status.tokens ?? null,
|
|
118
|
+
stories,
|
|
119
|
+
srcLoc: loc(join(runDir, 'src')),
|
|
120
|
+
}
|
|
121
|
+
validateResult(result)
|
|
122
|
+
|
|
123
|
+
mkdirSync(join(benchDir, 'results'), { recursive: true })
|
|
124
|
+
const out = join(benchDir, 'results', `${runner}-${stamp}.json`)
|
|
125
|
+
writeFileSync(out, JSON.stringify(result, null, 2) + '\n')
|
|
126
|
+
console.error(`[bench] done: ${out}`)
|
|
127
|
+
console.log(JSON.stringify(result, null, 2))
|
package/canon/AGENTS.md
CHANGED
|
@@ -17,6 +17,8 @@ When several skills could match the same task, resolve deterministically:
|
|
|
17
17
|
`tdd`, `subagent-driven-development`, `systematic-debugging`, …) take precedence and set the
|
|
18
18
|
process. Role skills (`review`, `ship`, `health`, `retro`, …) add a perspective on top.
|
|
19
19
|
2. **One canonical entrypoint per concern** — pick the most specific:
|
|
20
|
+
- Set up or update Yoke → `yoke-retrofit`
|
|
21
|
+
- Yoke-owned planning + autonomous story execution → `yoke-workflow`
|
|
20
22
|
- Plan-time architecture review → `plan-eng-review`
|
|
21
23
|
- Plan-time product / scope review → `plan-ceo-review`
|
|
22
24
|
- **Pre-merge code review → `review`** (the single canonical one)
|
package/canon/loop/loop-spec.md
CHANGED
|
@@ -4,7 +4,8 @@ The autonomous loop is OPTIONAL and toggle-able:
|
|
|
4
4
|
|
|
5
5
|
- `yoke loop on` / `yoke loop off` — enable/disable (recorded in `.yoke/config.yaml`, default off).
|
|
6
6
|
- `yoke loop status` — show enabled state + PRD progress.
|
|
7
|
-
- `yoke loop run [--max=N] [--isolate]` — run the loop (default cap 25 iterations).
|
|
7
|
+
- `yoke loop run [--max=N] [--isolate] [--decision-policy=auto|critical]` — run the loop (default cap 25 iterations).
|
|
8
|
+
- `yoke loop decision` / `yoke loop answer --choice=<id>` — inspect and answer a structured critical stop; answering records a human-owned, decision-file-only commit and resumes by default with the original runner, isolation, review, permission, timeout, and policy settings. `yoke loop resume` retries a restart that could not begin without weakening those settings.
|
|
8
9
|
|
|
9
10
|
Pass `--isolate` to run each iteration in a fresh git worktree: the agent works on a throwaway checkout, and only a verified, committed story is fast-forwarded back into the main tree. A failed iteration never touches your working tree. Requires `.yoke/prd.yaml` to be committed to git, since the worktree is a checkout of HEAD.
|
|
10
11
|
|
|
@@ -17,7 +18,8 @@ When enabled and run, each iteration:
|
|
|
17
18
|
1. Pre-dispatch gate: the git worktree must be clean, else `blocked`.
|
|
18
19
|
2. Pick the highest-priority unfinished PRD story (`.yoke/prd.yaml`).
|
|
19
20
|
3. Stop-the-Line gate: the story must have acceptance criteria, else `blocked`.
|
|
20
|
-
4. Run a fresh agent to implement ONE story.
|
|
21
|
+
4. Run a fresh agent to implement ONE story. Runner precedence is explicit `--runner`, configured `runner.agent`, active agent host, then the first configured agent. The loop refuses to start if that CLI is not installed.
|
|
22
|
+
With `decisionPolicy: auto`, routine ambiguity is resolved from the approved plan and project conventions. With `critical`, only high-impact architecture, security/privacy, destructive data, material-cost, compliance, or irreversible choices may produce `.yoke/decision-request.yaml`; the loop validates its bounded single-line fields, unique options, and active story ID, blocks before verify, and preserves it for `yoke loop answer`.
|
|
21
23
|
5. Run the project's verify command (config `verify.command`, or detected `npm test`).
|
|
22
24
|
**Verify is the source of truth** — the agent's exit code is advisory, so a spurious
|
|
23
25
|
non-zero exit (e.g. a Windows `.cmd` wrapper) cannot block a story whose tests are green.
|
package/canon/loop/prd.schema.md
CHANGED
|
@@ -6,9 +6,14 @@ The loop is driven by a versioned PRD file. Each story:
|
|
|
6
6
|
- id: STORY-1
|
|
7
7
|
title: Short imperative description
|
|
8
8
|
priority: 1 # lower = higher priority
|
|
9
|
+
needs: [] # optional dependency IDs; no unknown IDs, self-links, or cycles
|
|
10
|
+
area: api # optional collision domain for parallel scheduling
|
|
11
|
+
agent: codex # optional claude|codex|gemini affinity
|
|
9
12
|
acceptance: # Definition of Done (required before implementation)
|
|
10
13
|
- The endpoint returns 200 for a valid request.
|
|
11
14
|
passes: false # set true only when acceptance is met and tests are green
|
|
12
15
|
```
|
|
13
16
|
|
|
14
17
|
Stop condition: every story has `passes: true`.
|
|
18
|
+
|
|
19
|
+
Stories without `needs`, `area`, or `agent` retain the serial pre-1.0 behavior. A story is ready only when every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously active areas, and uses `agent` as an affinity hint.
|
package/canon/manifest.yaml
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
name: yoke-canon
|
|
2
|
-
version:
|
|
2
|
+
version: 1.1.0
|
|
3
3
|
agents: [claude, codex, gemini]
|
|
4
4
|
skills:
|
|
5
5
|
- { id: tdd, path: skills/tdd, kind: methodology }
|
|
6
6
|
- { id: yoke-retrofit, path: skills/yoke-retrofit, kind: methodology }
|
|
7
|
+
- { id: yoke-workflow, path: skills/yoke-workflow, kind: methodology }
|
|
7
8
|
- { id: minimal-code, path: skills/minimal-code, kind: methodology }
|
|
8
9
|
- { id: maintaining-context, path: skills/maintaining-context, kind: methodology }
|
|
9
10
|
# superpowers skills (kind: methodology)
|
|
@@ -26,9 +26,13 @@ good stories (small, testable, ordered) let it run overnight.
|
|
|
26
26
|
criterion. If the whole project has a budget, wire `perf.command` in `.yoke/config.yaml`
|
|
27
27
|
(see the `performance` skill) instead of repeating it per story.
|
|
28
28
|
7. **Ask everything now.** Clarifying questions belong in this planning round — a loop run
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
29
|
+
is unattended. A criterion that still contains `TBD` or another placeholder is not
|
|
30
|
+
loop-ready, and `yoke prd check` rejects it. During implementation,
|
|
31
|
+
`loop.decisionPolicy: auto` resolves routine ambiguity; `critical` pauses only for
|
|
32
|
+
high-impact choices and resumes after `yoke loop answer` records the answer.
|
|
33
|
+
8. **Model real dependencies.** Add `needs` only for hard prerequisites, `area` for files or
|
|
34
|
+
subsystems that must not be edited concurrently, and `agent` only as an affinity hint.
|
|
35
|
+
Dependency IDs must exist; self-dependencies and cycles are invalid.
|
|
32
36
|
|
|
33
37
|
## Format (`.yoke/prd.yaml`)
|
|
34
38
|
|
|
@@ -43,6 +47,9 @@ good stories (small, testable, ordered) let it run overnight.
|
|
|
43
47
|
- id: STORY-2
|
|
44
48
|
title: add the sum command
|
|
45
49
|
priority: 2
|
|
50
|
+
needs: [STORY-1]
|
|
51
|
+
area: cli
|
|
52
|
+
agent: codex
|
|
46
53
|
acceptance:
|
|
47
54
|
- "cli sum 1 2 prints 3"
|
|
48
55
|
- "non-numeric input exits 1 with an error message"
|
|
@@ -526,15 +526,10 @@ Analyze the diff and group changes into logical commits. Each commit should repr
|
|
|
526
526
|
|
|
527
527
|
**Each commit must be independently valid** — no broken imports, no references to code that doesn't exist yet.
|
|
528
528
|
|
|
529
|
-
The **final commit**
|
|
529
|
+
The **final commit** contains VERSION + CHANGELOG. The project's commit identity and co-author policy always wins; never add an AI co-author trailer unless the project explicitly allows it.
|
|
530
530
|
|
|
531
531
|
```bash
|
|
532
|
-
git commit -m "
|
|
533
|
-
chore: bump version and changelog (vX.Y.Z.W)
|
|
534
|
-
|
|
535
|
-
Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
|
|
536
|
-
EOF
|
|
537
|
-
)"
|
|
532
|
+
git commit -m "chore: bump version and changelog (vX.Y.Z.W)"
|
|
538
533
|
```
|
|
539
534
|
|
|
540
535
|
---
|
|
@@ -7,6 +7,10 @@ description: Use at the start of any non-trivial task — the default order of o
|
|
|
7
7
|
|
|
8
8
|
For any non-trivial change, move through these phases in order (skip only what genuinely does not apply):
|
|
9
9
|
|
|
10
|
+
When `.yoke/config.yaml` exists and the user wants Yoke to own planning plus autonomous
|
|
11
|
+
execution, use `yoke-workflow` as the entrypoint. It adds the approved-plan → PRD → loop
|
|
12
|
+
handoff and the configured critical-decision behavior to the phases below.
|
|
13
|
+
|
|
10
14
|
1. **Brainstorm** the idea into a clear design — see `brainstorming`.
|
|
11
15
|
2. **Plan** a concrete, testable implementation — see `writing-plans`.
|
|
12
16
|
3. **Understand the code** — map the blast radius with the code-graph before changing anything.
|
|
@@ -1,19 +1,26 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: yoke-retrofit
|
|
3
|
-
description: Use when asked to "retrofit", "yoke this project", or set up the Yoke harness in a project — runs
|
|
3
|
+
description: Use when asked to "retrofit", "yoke this project", or set up the Yoke harness in a project — runs the shared setup wizard and configures the same behavior for Claude, Codex, and Gemini.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Yoke Retrofit
|
|
7
7
|
|
|
8
|
-
Set up
|
|
8
|
+
Set up or update Yoke through the shared `yoke setup` contract.
|
|
9
9
|
|
|
10
|
-
1.
|
|
11
|
-
|
|
12
|
-
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
10
|
+
1. Inspect the project and identify the current host (`claude`, `codex`, or `gemini`).
|
|
11
|
+
2. Ask these setup questions one at a time and give a direct recommendation:
|
|
12
|
+
- target agents (recommend the current host; use `all` for deliberately cross-agent projects),
|
|
13
|
+
- code-graph tool,
|
|
14
|
+
- autonomous loop on/off,
|
|
15
|
+
- default runner (recommend the current host),
|
|
16
|
+
- decision mode: `auto` or `critical`.
|
|
17
|
+
3. Recommend the code graph based on this project:
|
|
18
|
+
- **Serena** is LSP-accurate and best for large typed codebases or systematic symbol refactors where a missed reference is costly. It needs a language server per language.
|
|
19
|
+
- **graphify** is fast and multimodal, and is best for exploration, migration, onboarding, or mixed code and document repositories. Its graph is an index and can become stale.
|
|
20
|
+
4. Apply the answers without a second round of prompts:
|
|
21
|
+
`yoke setup . --yes --host=<host> --agent=<agents> --code-graph=<choice> --runner=<runner> --decision-policy=<auto|critical> --loop|--no-loop`.
|
|
22
|
+
A human who runs `yoke setup .` directly receives the same five terminal questions.
|
|
23
|
+
5. Show the generated report and backup paths. Existing files are backed up under `.yoke/backup/`; settings are merged where supported.
|
|
24
|
+
6. If an old generated `CLAUDE.md` or `GEMINI.md` contained project-specific instructions, restore them inside its `<!-- yoke:preserve:start -->` / `<!-- yoke:preserve:end -->` block. Preserve blocks survive every later retrofit.
|
|
18
25
|
|
|
19
|
-
The harness includes
|
|
26
|
+
The generated harness includes the provider-neutral `yoke-workflow` skill. It owns the planning questions, approved-plan handoff, autonomous stories, and critical-decision resume flow.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: yoke-workflow
|
|
3
|
+
description: Use when the user asks Yoke to plan and build a feature, run stories autonomously, continue a Yoke loop, or only interrupt for major decisions.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Yoke Workflow
|
|
7
|
+
|
|
8
|
+
Provide the same interaction contract in Claude, Codex, and Gemini.
|
|
9
|
+
|
|
10
|
+
1. Read `.yoke/config.yaml`. If it is missing, offer `yoke setup . --host=<current-agent>` and run the setup flow before planning.
|
|
11
|
+
2. Plan before starting the loop. Inspect the project, then ask one focused question at a time only where the answer changes product behavior, scope, architecture, security, data ownership, external cost, or an irreversible choice. Include a recommended answer. Resolve routine implementation details yourself.
|
|
12
|
+
3. Summarize the agreed design in `.yoke/plan.md`, including goals, non-goals, constraints, and decisions. Use the `authoring-prd` skill to turn it into small stories with testable acceptance criteria. Run `yoke prd check .`.
|
|
13
|
+
4. Ask once for approval of the complete plan and story set. Do not begin implementation before that approval.
|
|
14
|
+
5. If `loop.enabled` is true, run the stories without routine follow-up questions using the configured runner. Prefer `yoke loop run . --max=5 --isolate`, report status after each batch, and continue until complete or genuinely blocked.
|
|
15
|
+
6. Respect `loop.decisionPolicy`:
|
|
16
|
+
- `auto`: choose the most suitable option from the plan, current code, and established conventions. Record the interpretation and continue.
|
|
17
|
+
- `critical`: routine ambiguity is still resolved automatically. If the loop reports a pending critical decision, run `yoke loop decision .`, present its options and recommendation to the user, ask exactly that question, then run `yoke loop answer . --choice=<id> --rationale="<answer>"`. The answer command records the decision and resumes the same story.
|
|
18
|
+
7. Never ask whether to run tests, review, commit, or continue to the next approved story. Those are part of the approved workflow.
|
|
19
|
+
|
|
20
|
+
The user's configured commit identity is authoritative. Do not add an AI co-author unless `commit.allowCoAuthors` explicitly permits it.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { spawnSync } from 'node:child_process'
|
|
2
|
+
import { resolve } from 'node:path'
|
|
3
|
+
import { pathToFileURL } from 'node:url'
|
|
4
|
+
|
|
5
|
+
function rtkCheck(command) {
|
|
6
|
+
const result = spawnSync('rtk', ['hook', 'check', command], { encoding: 'utf8', timeout: 3000 })
|
|
7
|
+
return result.status === 0 ? result.stdout.trim() : ''
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export function rewriteHookInput(input, check = rtkCheck) {
|
|
11
|
+
if (input?.tool_name !== 'Bash' && input?.toolName !== 'Bash') return null
|
|
12
|
+
const toolInput = input.tool_input ?? input.toolInput
|
|
13
|
+
const command = toolInput?.command
|
|
14
|
+
if (typeof command !== 'string' || command.trim() === '') return null
|
|
15
|
+
const rewritten = check(command)
|
|
16
|
+
if (!rewritten || rewritten === command) return null
|
|
17
|
+
return {
|
|
18
|
+
hookSpecificOutput: {
|
|
19
|
+
hookEventName: 'PreToolUse',
|
|
20
|
+
updatedInput: { ...toolInput, command: rewritten },
|
|
21
|
+
},
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
async function main() {
|
|
26
|
+
let raw = ''
|
|
27
|
+
for await (const chunk of process.stdin) raw += chunk
|
|
28
|
+
try {
|
|
29
|
+
const output = rewriteHookInput(JSON.parse(raw))
|
|
30
|
+
if (output) process.stdout.write(JSON.stringify(output))
|
|
31
|
+
} catch {
|
|
32
|
+
// Compression is an optimization. Malformed input must never block Codex.
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
if (process.argv[1] && pathToFileURL(resolve(process.argv[1])).href === import.meta.url) await main()
|