@hecer/yoke 0.9.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codex-plugin/plugin.json +7 -0
- package/CHANGELOG.md +169 -149
- package/README.md +24 -16
- package/TODOS.md +8 -0
- package/agents/docs.toml +6 -0
- package/agents/implementer.toml +6 -0
- package/agents/reviewer.toml +6 -0
- package/agents/security.toml +6 -0
- package/bench/README.md +45 -42
- package/bench/RESULTS.md +46 -36
- package/bench/result-schema.mjs +12 -0
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
- package/bench/results/codex-unavailable-1785175418318.json +15 -0
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
- package/bench/run-matrix.mjs +26 -0
- package/bench/run.mjs +127 -115
- package/canon/loop/prd.schema.md +5 -0
- package/canon/manifest.yaml +1 -1
- package/canon/skills/authoring-prd/SKILL.md +6 -0
- package/canon/skills/ship/SKILL.md +2 -7
- package/canon/tools/codex-rtk-hook.mjs +36 -0
- package/dist/agents/providers.js +23 -0
- package/dist/agents/telemetry.js +30 -0
- package/dist/agents/types.js +1 -0
- package/dist/audit/changes.js +6 -0
- package/dist/audit/command.js +64 -0
- package/dist/audit/dependencies.js +21 -0
- package/dist/audit/secrets.js +16 -0
- package/dist/audit/types.js +1 -0
- package/dist/cli.js +22 -4
- package/dist/loop/claims.js +57 -0
- package/dist/loop/cleanup.js +10 -4
- package/dist/loop/git.js +8 -2
- package/dist/loop/identity.js +27 -0
- package/dist/loop/loop.js +20 -2
- package/dist/loop/merge-queue.js +20 -0
- package/dist/loop/parallel.js +39 -0
- package/dist/loop/prd.js +48 -2
- package/dist/loop/run-command.js +51 -6
- package/dist/loop/runner.js +41 -29
- package/dist/loop/scheduler.js +8 -0
- package/dist/prd/command.js +6 -0
- package/dist/retrofit/config.js +12 -0
- package/dist/retrofit/planners/codex.js +64 -19
- package/dist/review/command.js +52 -12
- package/dist/review/verdict.js +45 -0
- package/docs/MIGRATING-TO-1.0.md +33 -0
- package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
- package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
- package/hooks/hooks.json +19 -0
- package/package.json +82 -67
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
- package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
- package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
- package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
- package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
- package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
package/bench/RESULTS.md
CHANGED
|
@@ -1,36 +1,46 @@
|
|
|
1
|
-
# Benchmark results
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
1
|
+
# Benchmark results
|
|
2
|
+
|
|
3
|
+
Result schema v1 records fixture version, sample label, permission profile, telemetry/model
|
|
4
|
+
availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
|
|
5
|
+
|
|
6
|
+
Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
|
|
7
|
+
[README.md](README.md). One row per run — raw JSON in [`results/`](results/).
|
|
8
|
+
|
|
9
|
+
## Runs
|
|
10
|
+
|
|
11
|
+
| Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
|
|
12
|
+
|---|---|---|---|---|---|---|---|---|
|
|
13
|
+
| 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
|
|
14
|
+
| 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
|
|
15
|
+
| 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
|
|
16
|
+
| 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
|
|
17
|
+
| 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
|
|
18
|
+
| — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
|
|
19
|
+
|
|
20
|
+
Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
|
|
21
|
+
passed verify on the first iteration; the final quality check (all 16 assertions on the final
|
|
22
|
+
tree, outside the loop) is green.
|
|
23
|
+
|
|
24
|
+
The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
|
|
25
|
+
before changing the fixture, and the Codex executable could not be probed in this Windows
|
|
26
|
+
environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
|
|
27
|
+
|
|
28
|
+
## What the harness caught before producing a single number
|
|
29
|
+
|
|
30
|
+
Building an honest benchmark is itself a verification pass. The first runs found two real
|
|
31
|
+
Yoke bugs, both fixed in 0.3.0:
|
|
32
|
+
|
|
33
|
+
1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
|
|
34
|
+
5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
|
|
35
|
+
2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
|
|
36
|
+
`-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
|
|
37
|
+
stdin (which selects headless mode) and passes only `--yolo`.
|
|
38
|
+
|
|
39
|
+
## Reading the numbers
|
|
40
|
+
|
|
41
|
+
- Tokens come from the loop's own hook (claude runner only — gemini/codex reporting is an
|
|
42
|
+
open gap, see the multi-agent design doc).
|
|
43
|
+
- N=1: indicative, not statistics. Re-run with `node bench/run.mjs --runner=<agent>` and
|
|
44
|
+
compare rows.
|
|
45
|
+
- ~53 k tokens / ~4.5 minutes for a 3-story micro-backlog is the current price of the full
|
|
46
|
+
gate pipeline (fresh headless session per story + verify + atomic commit) on this fixture.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
export const requiredResultFields = [
|
|
2
|
+
'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
|
|
3
|
+
'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
|
|
4
|
+
'iterations', 'finalTestsPass',
|
|
5
|
+
]
|
|
6
|
+
|
|
7
|
+
export function validateResult(result) {
|
|
8
|
+
const missing = requiredResultFields.filter(key => !(key in result))
|
|
9
|
+
if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
|
|
10
|
+
if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
|
|
11
|
+
return result
|
|
12
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "claude",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"yokeVersion": "1.0.0",
|
|
8
|
+
"fixture": "string-kit",
|
|
9
|
+
"startedAt": "2026-07-27T18:03:26.598Z",
|
|
10
|
+
"wallClockMs": 11345,
|
|
11
|
+
"exitCode": 1,
|
|
12
|
+
"finalState": "blocked",
|
|
13
|
+
"verdict": "blocked",
|
|
14
|
+
"blocker": "Claude runner exited before implementation; fixture verification remained red.",
|
|
15
|
+
"conflicts": 0,
|
|
16
|
+
"iterations": 1,
|
|
17
|
+
"finalTestsPass": false,
|
|
18
|
+
"progress": {
|
|
19
|
+
"passed": 0,
|
|
20
|
+
"total": 3
|
|
21
|
+
},
|
|
22
|
+
"usageAvailable": false,
|
|
23
|
+
"modelAvailable": false,
|
|
24
|
+
"tokens": {
|
|
25
|
+
"inputTokens": 0,
|
|
26
|
+
"outputTokens": 0,
|
|
27
|
+
"model": "<synthetic>"
|
|
28
|
+
},
|
|
29
|
+
"stories": [
|
|
30
|
+
{
|
|
31
|
+
"id": "STORY-1",
|
|
32
|
+
"durationMs": 10954,
|
|
33
|
+
"iterations": 1,
|
|
34
|
+
"finalTestsPass": false
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"id": "STORY-2",
|
|
38
|
+
"durationMs": null,
|
|
39
|
+
"iterations": 0,
|
|
40
|
+
"finalTestsPass": false
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"id": "STORY-3",
|
|
44
|
+
"durationMs": null,
|
|
45
|
+
"iterations": 0,
|
|
46
|
+
"finalTestsPass": false
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"srcLoc": 3
|
|
50
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "codex",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"usageAvailable": false,
|
|
8
|
+
"modelAvailable": false,
|
|
9
|
+
"verdict": "unavailable",
|
|
10
|
+
"blocker": "Zugriff verweigert",
|
|
11
|
+
"conflicts": 0,
|
|
12
|
+
"wallClockMs": null,
|
|
13
|
+
"iterations": 0,
|
|
14
|
+
"finalTestsPass": false
|
|
15
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "gemini",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"yokeVersion": "1.0.0",
|
|
8
|
+
"fixture": "string-kit",
|
|
9
|
+
"startedAt": "2026-07-27T18:03:45.223Z",
|
|
10
|
+
"wallClockMs": 13844,
|
|
11
|
+
"exitCode": 1,
|
|
12
|
+
"finalState": "blocked",
|
|
13
|
+
"verdict": "blocked",
|
|
14
|
+
"blocker": "Gemini runner exited before implementation; fixture verification remained red.",
|
|
15
|
+
"conflicts": 0,
|
|
16
|
+
"iterations": 1,
|
|
17
|
+
"finalTestsPass": false,
|
|
18
|
+
"progress": {
|
|
19
|
+
"passed": 0,
|
|
20
|
+
"total": 3
|
|
21
|
+
},
|
|
22
|
+
"usageAvailable": false,
|
|
23
|
+
"modelAvailable": false,
|
|
24
|
+
"tokens": null,
|
|
25
|
+
"stories": [
|
|
26
|
+
{
|
|
27
|
+
"id": "STORY-1",
|
|
28
|
+
"durationMs": 7040,
|
|
29
|
+
"iterations": 1,
|
|
30
|
+
"finalTestsPass": false
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"id": "STORY-2",
|
|
34
|
+
"durationMs": null,
|
|
35
|
+
"iterations": 0,
|
|
36
|
+
"finalTestsPass": false
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"id": "STORY-3",
|
|
40
|
+
"durationMs": null,
|
|
41
|
+
"iterations": 0,
|
|
42
|
+
"finalTestsPass": false
|
|
43
|
+
}
|
|
44
|
+
],
|
|
45
|
+
"srcLoc": 3
|
|
46
|
+
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { spawnSync } from 'node:child_process'
|
|
3
|
+
import { mkdirSync, writeFileSync } from 'node:fs'
|
|
4
|
+
import { dirname, join } from 'node:path'
|
|
5
|
+
import { fileURLToPath } from 'node:url'
|
|
6
|
+
import { validateResult } from './result-schema.mjs'
|
|
7
|
+
|
|
8
|
+
const benchDir = dirname(fileURLToPath(import.meta.url))
|
|
9
|
+
if (process.argv.includes('--help')) {
|
|
10
|
+
console.log('usage: node bench/run-matrix.mjs [--label=sample]')
|
|
11
|
+
process.exit(0)
|
|
12
|
+
}
|
|
13
|
+
const label = process.argv.find(arg => arg.startsWith('--label='))?.slice(8) ?? `matrix-${new Date().toISOString()}`
|
|
14
|
+
mkdirSync(join(benchDir, 'results'), { recursive: true })
|
|
15
|
+
for (const runner of ['claude', 'codex', 'gemini']) {
|
|
16
|
+
const probe = spawnSync(runner, ['--version'], { encoding: 'utf8', shell: process.platform === 'win32', timeout: 20_000 })
|
|
17
|
+
if (probe.status !== 0) {
|
|
18
|
+
const row = validateResult({ schemaVersion: 1, fixtureVersion: 'string-kit@1', runner, sampleLabel: label, permissionProfile: 'safe', usageAvailable: false, modelAvailable: false, verdict: 'unavailable', blocker: (probe.stderr || probe.error?.message || 'CLI unavailable').trim(), conflicts: 0, wallClockMs: null, iterations: 0, finalTestsPass: false })
|
|
19
|
+
const out = join(benchDir, 'results', `${runner}-unavailable-${Date.now()}.json`)
|
|
20
|
+
writeFileSync(out, JSON.stringify(row, null, 2) + '\n')
|
|
21
|
+
console.log(JSON.stringify(row))
|
|
22
|
+
continue
|
|
23
|
+
}
|
|
24
|
+
const run = spawnSync(process.execPath, [join(benchDir, 'run.mjs'), `--runner=${runner}`, `--label=${label}`], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'inherit'] })
|
|
25
|
+
if (run.stdout) process.stdout.write(run.stdout)
|
|
26
|
+
}
|
package/bench/run.mjs
CHANGED
|
@@ -1,115 +1,127 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
// Yoke benchmark harness.
|
|
3
|
-
//
|
|
4
|
-
// node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
|
|
5
|
-
//
|
|
6
|
-
// Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
|
|
7
|
-
// `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
|
|
8
|
-
// durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
|
|
9
|
-
// the loop's token hook (claude runner only), and quality as the fixture's own
|
|
10
|
-
// pre-written tests — run per story AFTER the loop finishes, on the final tree.
|
|
11
|
-
import { spawn, spawnSync } from 'node:child_process'
|
|
12
|
-
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
|
|
13
|
-
import { join, dirname } from 'node:path'
|
|
14
|
-
import { fileURLToPath } from 'node:url'
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
const
|
|
18
|
-
const
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
)
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
const
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
const
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
git('
|
|
45
|
-
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', '
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
const
|
|
54
|
-
const
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
const
|
|
74
|
-
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
const
|
|
78
|
-
const
|
|
79
|
-
const
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Yoke benchmark harness.
|
|
3
|
+
//
|
|
4
|
+
// node bench/run.mjs --runner=claude [--max=6] [--timeout=10] [--label=note]
|
|
5
|
+
//
|
|
6
|
+
// Copies the fixture into bench/.runs/<runner>-<stamp>, git-inits it, then drives
|
|
7
|
+
// `yoke loop run --json` and measures from the OUTSIDE (the loop itself records no
|
|
8
|
+
// durations): per-story wall-clock from NDJSON event timestamps, tokens/model from
|
|
9
|
+
// the loop's token hook (claude runner only), and quality as the fixture's own
|
|
10
|
+
// pre-written tests — run per story AFTER the loop finishes, on the final tree.
|
|
11
|
+
import { spawn, spawnSync } from 'node:child_process'
|
|
12
|
+
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync } from 'node:fs'
|
|
13
|
+
import { join, dirname } from 'node:path'
|
|
14
|
+
import { fileURLToPath } from 'node:url'
|
|
15
|
+
import { validateResult } from './result-schema.mjs'
|
|
16
|
+
|
|
17
|
+
const benchDir = dirname(fileURLToPath(import.meta.url))
|
|
18
|
+
const repoRoot = dirname(benchDir)
|
|
19
|
+
const cli = join(repoRoot, 'dist', 'cli.js')
|
|
20
|
+
|
|
21
|
+
const args = Object.fromEntries(
|
|
22
|
+
process.argv.slice(2).filter(a => a.startsWith('--')).map(a => {
|
|
23
|
+
const [k, v] = a.slice(2).split('=')
|
|
24
|
+
return [k, v ?? true]
|
|
25
|
+
}),
|
|
26
|
+
)
|
|
27
|
+
const runner = args.runner
|
|
28
|
+
if (!['claude', 'codex', 'gemini'].includes(runner)) {
|
|
29
|
+
console.error('usage: node bench/run.mjs --runner=<claude|codex|gemini> [--max=6] [--timeout=10] [--label=note]')
|
|
30
|
+
process.exit(2)
|
|
31
|
+
}
|
|
32
|
+
const max = Number(args.max ?? 6)
|
|
33
|
+
const timeout = Number(args.timeout ?? 10)
|
|
34
|
+
|
|
35
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
|
|
36
|
+
const runDir = join(benchDir, '.runs', `${runner}-${stamp}`)
|
|
37
|
+
mkdirSync(runDir, { recursive: true })
|
|
38
|
+
cpSync(join(benchDir, 'fixtures', 'string-kit'), runDir, { recursive: true })
|
|
39
|
+
|
|
40
|
+
const git = (...a) => {
|
|
41
|
+
const r = spawnSync('git', ['-C', runDir, ...a], { encoding: 'utf8' })
|
|
42
|
+
if (r.status !== 0) throw new Error(`git ${a.join(' ')} failed: ${r.stderr}`)
|
|
43
|
+
}
|
|
44
|
+
git('init', '-q')
|
|
45
|
+
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'add', '-A')
|
|
46
|
+
git('-c', 'user.name=bench', '-c', 'user.email=bench@yoke', 'commit', '-q', '-m', 'bench: fixture baseline')
|
|
47
|
+
|
|
48
|
+
// A nested Claude Code session refuses some operations; scrub session markers.
|
|
49
|
+
const env = { ...process.env }
|
|
50
|
+
for (const k of Object.keys(env)) if (k.startsWith('CLAUDE_CODE') || k === 'CLAUDECODE') delete env[k]
|
|
51
|
+
|
|
52
|
+
console.error(`[bench] ${runner} → ${runDir}`)
|
|
53
|
+
const t0 = Date.now()
|
|
54
|
+
const events = []
|
|
55
|
+
const child = spawn(process.execPath, [cli, 'loop', 'run', runDir, '--json', `--runner=${runner}`, `--max=${max}`, `--timeout=${timeout}`], {
|
|
56
|
+
env, stdio: ['ignore', 'pipe', 'inherit'],
|
|
57
|
+
})
|
|
58
|
+
let buf = ''
|
|
59
|
+
child.stdout.on('data', d => {
|
|
60
|
+
buf += d
|
|
61
|
+
let i
|
|
62
|
+
while ((i = buf.indexOf('\n')) >= 0) {
|
|
63
|
+
const line = buf.slice(0, i).trim()
|
|
64
|
+
buf = buf.slice(i + 1)
|
|
65
|
+
if (!line) continue
|
|
66
|
+
try { events.push({ at: Date.now(), ...JSON.parse(line) }) } catch { /* non-JSON noise */ }
|
|
67
|
+
}
|
|
68
|
+
})
|
|
69
|
+
const exitCode = await new Promise(res => child.on('close', res))
|
|
70
|
+
const wallClockMs = Date.now() - t0
|
|
71
|
+
|
|
72
|
+
// Per-story duration: first event mentioning the story -> first event mentioning the next story (or end).
|
|
73
|
+
const storyIds = ['STORY-1', 'STORY-2', 'STORY-3']
|
|
74
|
+
const firstSeen = {}
|
|
75
|
+
for (const e of events) if (e.story && !(e.story in firstSeen)) firstSeen[e.story] = e.at
|
|
76
|
+
const stories = storyIds.map((id, idx) => {
|
|
77
|
+
const start = firstSeen[id]
|
|
78
|
+
const next = storyIds.slice(idx + 1).map(n => firstSeen[n]).find(v => v !== undefined)
|
|
79
|
+
const durationMs = start === undefined ? null : (next ?? t0 + wallClockMs) - start
|
|
80
|
+
const iterations = new Set(events.filter(e => e.story === id).map(e => e.iteration)).size
|
|
81
|
+
// Quality: the fixture's own tests for this story, on the final tree.
|
|
82
|
+
const q = spawnSync(process.execPath, ['--test', `tests/${id}.test.mjs`], { cwd: runDir, encoding: 'utf8' })
|
|
83
|
+
return { id, durationMs, iterations, finalTestsPass: q.status === 0 }
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
const last = events[events.length - 1] ?? {}
|
|
87
|
+
let status = {}
|
|
88
|
+
try { status = JSON.parse(readFileSync(join(runDir, '.yoke', 'loop-status.json'), 'utf8')) } catch { /* loop may have refused before writing status */ }
|
|
89
|
+
|
|
90
|
+
// Source size (LOC in src/) as a code-economy proxy.
|
|
91
|
+
const loc = (dir) => readdirSync(dir).reduce((n, f) => {
|
|
92
|
+
const p = join(dir, f)
|
|
93
|
+
if (statSync(p).isDirectory()) return n + loc(p)
|
|
94
|
+
return n + readFileSync(p, 'utf8').split('\n').filter(l => l.trim() !== '').length
|
|
95
|
+
}, 0)
|
|
96
|
+
|
|
97
|
+
const result = {
|
|
98
|
+
schemaVersion: 1,
|
|
99
|
+
fixtureVersion: 'string-kit@1',
|
|
100
|
+
runner,
|
|
101
|
+
sampleLabel: String(args.label ?? `${runner}-${stamp}`),
|
|
102
|
+
permissionProfile: args.unsafe ? 'unsafe' : 'safe',
|
|
103
|
+
yokeVersion: JSON.parse(readFileSync(join(repoRoot, 'package.json'), 'utf8')).version,
|
|
104
|
+
fixture: 'string-kit',
|
|
105
|
+
startedAt: new Date(t0).toISOString(),
|
|
106
|
+
wallClockMs,
|
|
107
|
+
exitCode,
|
|
108
|
+
finalState: last.state ?? null,
|
|
109
|
+
verdict: exitCode === 0 ? 'completed' : (/api key|login|auth/i.test(String(status.reason ?? '')) ? 'auth-failed' : 'blocked'),
|
|
110
|
+
blocker: exitCode === 0 ? null : (status.reason ?? 'runner exited without a diagnostic'),
|
|
111
|
+
conflicts: events.filter(e => /conflict/i.test(String(e.reason ?? e.summary ?? ''))).length,
|
|
112
|
+
iterations: stories.reduce((sum, story) => sum + story.iterations, 0),
|
|
113
|
+
finalTestsPass: stories.every(story => story.finalTestsPass),
|
|
114
|
+
progress: last.progress ?? null,
|
|
115
|
+
usageAvailable: Number(status.tokens?.inputTokens ?? 0) + Number(status.tokens?.outputTokens ?? 0) > 0,
|
|
116
|
+
modelAvailable: typeof status.tokens?.model === 'string' && status.tokens.model !== '<synthetic>',
|
|
117
|
+
tokens: status.tokens ?? null,
|
|
118
|
+
stories,
|
|
119
|
+
srcLoc: loc(join(runDir, 'src')),
|
|
120
|
+
}
|
|
121
|
+
validateResult(result)
|
|
122
|
+
|
|
123
|
+
mkdirSync(join(benchDir, 'results'), { recursive: true })
|
|
124
|
+
const out = join(benchDir, 'results', `${runner}-${stamp}.json`)
|
|
125
|
+
writeFileSync(out, JSON.stringify(result, null, 2) + '\n')
|
|
126
|
+
console.error(`[bench] done: ${out}`)
|
|
127
|
+
console.log(JSON.stringify(result, null, 2))
|
package/canon/loop/prd.schema.md
CHANGED
|
@@ -6,9 +6,14 @@ The loop is driven by a versioned PRD file. Each story:
|
|
|
6
6
|
- id: STORY-1
|
|
7
7
|
title: Short imperative description
|
|
8
8
|
priority: 1 # lower = higher priority
|
|
9
|
+
needs: [] # optional dependency IDs; no unknown IDs, self-links, or cycles
|
|
10
|
+
area: api # optional collision domain for parallel scheduling
|
|
11
|
+
agent: codex # optional claude|codex|gemini affinity
|
|
9
12
|
acceptance: # Definition of Done (required before implementation)
|
|
10
13
|
- The endpoint returns 200 for a valid request.
|
|
11
14
|
passes: false # set true only when acceptance is met and tests are green
|
|
12
15
|
```
|
|
13
16
|
|
|
14
17
|
Stop condition: every story has `passes: true`.
|
|
18
|
+
|
|
19
|
+
Stories without `needs`, `area`, or `agent` retain the serial pre-1.0 behavior. A story is ready only when every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously active areas, and uses `agent` as an affinity hint.
|
package/canon/manifest.yaml
CHANGED
|
@@ -29,6 +29,9 @@ good stories (small, testable, ordered) let it run overnight.
|
|
|
29
29
|
has nobody to ask. A criterion that still needs a decision ("TBD", "choose a provider")
|
|
30
30
|
is not loop-ready; resolve it here or the agent will either guess (default) or block
|
|
31
31
|
(`--on-ambiguity=abort`).
|
|
32
|
+
8. **Model real dependencies.** Add `needs` only for hard prerequisites, `area` for files or
|
|
33
|
+
subsystems that must not be edited concurrently, and `agent` only as an affinity hint.
|
|
34
|
+
Dependency IDs must exist; self-dependencies and cycles are invalid.
|
|
32
35
|
|
|
33
36
|
## Format (`.yoke/prd.yaml`)
|
|
34
37
|
|
|
@@ -43,6 +46,9 @@ good stories (small, testable, ordered) let it run overnight.
|
|
|
43
46
|
- id: STORY-2
|
|
44
47
|
title: add the sum command
|
|
45
48
|
priority: 2
|
|
49
|
+
needs: [STORY-1]
|
|
50
|
+
area: cli
|
|
51
|
+
agent: codex
|
|
46
52
|
acceptance:
|
|
47
53
|
- "cli sum 1 2 prints 3"
|
|
48
54
|
- "non-numeric input exits 1 with an error message"
|
|
@@ -526,15 +526,10 @@ Analyze the diff and group changes into logical commits. Each commit should repr
|
|
|
526
526
|
|
|
527
527
|
**Each commit must be independently valid** — no broken imports, no references to code that doesn't exist yet.
|
|
528
528
|
|
|
529
|
-
The **final commit**
|
|
529
|
+
The **final commit** contains VERSION + CHANGELOG. The project's commit identity and co-author policy always wins; never add an AI co-author trailer unless the project explicitly allows it.
|
|
530
530
|
|
|
531
531
|
```bash
|
|
532
|
-
git commit -m "
|
|
533
|
-
chore: bump version and changelog (vX.Y.Z.W)
|
|
534
|
-
|
|
535
|
-
Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
|
|
536
|
-
EOF
|
|
537
|
-
)"
|
|
532
|
+
git commit -m "chore: bump version and changelog (vX.Y.Z.W)"
|
|
538
533
|
```
|
|
539
534
|
|
|
540
535
|
---
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { spawnSync } from 'node:child_process'
|
|
2
|
+
import { resolve } from 'node:path'
|
|
3
|
+
import { pathToFileURL } from 'node:url'
|
|
4
|
+
|
|
5
|
+
function rtkCheck(command) {
|
|
6
|
+
const result = spawnSync('rtk', ['hook', 'check', command], { encoding: 'utf8', timeout: 3000 })
|
|
7
|
+
return result.status === 0 ? result.stdout.trim() : ''
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export function rewriteHookInput(input, check = rtkCheck) {
|
|
11
|
+
if (input?.tool_name !== 'Bash' && input?.toolName !== 'Bash') return null
|
|
12
|
+
const toolInput = input.tool_input ?? input.toolInput
|
|
13
|
+
const command = toolInput?.command
|
|
14
|
+
if (typeof command !== 'string' || command.trim() === '') return null
|
|
15
|
+
const rewritten = check(command)
|
|
16
|
+
if (!rewritten || rewritten === command) return null
|
|
17
|
+
return {
|
|
18
|
+
hookSpecificOutput: {
|
|
19
|
+
hookEventName: 'PreToolUse',
|
|
20
|
+
updatedInput: { ...toolInput, command: rewritten },
|
|
21
|
+
},
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
async function main() {
|
|
26
|
+
let raw = ''
|
|
27
|
+
for await (const chunk of process.stdin) raw += chunk
|
|
28
|
+
try {
|
|
29
|
+
const output = rewriteHookInput(JSON.parse(raw))
|
|
30
|
+
if (output) process.stdout.write(JSON.stringify(output))
|
|
31
|
+
} catch {
|
|
32
|
+
// Compression is an optimization. Malformed input must never block Codex.
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
if (process.argv[1] && pathToFileURL(resolve(process.argv[1])).href === import.meta.url) await main()
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
const argsFor = (agent, permissions) => {
|
|
2
|
+
if (agent === 'claude') {
|
|
3
|
+
const mode = permissions === 'unsafe' ? 'bypassPermissions' : permissions === 'read-only' ? 'plan' : 'auto';
|
|
4
|
+
const args = ['-p', '--permission-mode', mode];
|
|
5
|
+
if (permissions === 'unsafe')
|
|
6
|
+
args.push('--dangerously-skip-permissions');
|
|
7
|
+
return [...args, '--output-format', 'stream-json', '--verbose'];
|
|
8
|
+
}
|
|
9
|
+
if (agent === 'codex') {
|
|
10
|
+
if (permissions === 'unsafe')
|
|
11
|
+
return ['exec', '--dangerously-bypass-approvals-and-sandbox', '--json'];
|
|
12
|
+
if (permissions === 'read-only')
|
|
13
|
+
return ['exec', '--sandbox', 'read-only', '--json'];
|
|
14
|
+
return ['exec', '--full-auto', '--json'];
|
|
15
|
+
}
|
|
16
|
+
if (permissions === 'unsafe')
|
|
17
|
+
return ['--yolo', '--output-format', 'stream-json'];
|
|
18
|
+
const approval = permissions === 'read-only' ? 'plan' : 'auto_edit';
|
|
19
|
+
return ['--approval-mode', approval, '--sandbox', '--output-format', 'stream-json'];
|
|
20
|
+
};
|
|
21
|
+
export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe') {
|
|
22
|
+
return { command: agent, args: argsFor(agent, permissions), input: prompt, cwd };
|
|
23
|
+
}
|