@hecer/yoke 0.8.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.codex-plugin/plugin.json +7 -0
  2. package/CHANGELOG.md +169 -130
  3. package/README.md +61 -21
  4. package/TODOS.md +8 -0
  5. package/agents/docs.toml +6 -0
  6. package/agents/implementer.toml +6 -0
  7. package/agents/reviewer.toml +6 -0
  8. package/agents/security.toml +6 -0
  9. package/bench/README.md +45 -42
  10. package/bench/RESULTS.md +46 -36
  11. package/bench/result-schema.mjs +12 -0
  12. package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
  13. package/bench/results/codex-unavailable-1785175418318.json +15 -0
  14. package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
  15. package/bench/run-matrix.mjs +26 -0
  16. package/bench/run.mjs +127 -115
  17. package/canon/loop/prd.schema.md +5 -0
  18. package/canon/manifest.yaml +3 -1
  19. package/canon/skills/authoring-prd/SKILL.md +14 -0
  20. package/canon/skills/performance/SKILL.md +48 -0
  21. package/canon/skills/ship/SKILL.md +2 -7
  22. package/canon/tools/codex-rtk-hook.mjs +36 -0
  23. package/dist/agents/providers.js +23 -0
  24. package/dist/agents/telemetry.js +30 -0
  25. package/dist/agents/types.js +1 -0
  26. package/dist/audit/changes.js +6 -0
  27. package/dist/audit/command.js +64 -0
  28. package/dist/audit/dependencies.js +21 -0
  29. package/dist/audit/secrets.js +16 -0
  30. package/dist/audit/types.js +1 -0
  31. package/dist/cli.js +22 -4
  32. package/dist/loop/claims.js +57 -0
  33. package/dist/loop/cleanup.js +10 -4
  34. package/dist/loop/git.js +8 -2
  35. package/dist/loop/identity.js +27 -0
  36. package/dist/loop/loop.js +55 -26
  37. package/dist/loop/merge-queue.js +20 -0
  38. package/dist/loop/parallel.js +39 -0
  39. package/dist/loop/prd.js +48 -2
  40. package/dist/loop/run-command.js +63 -7
  41. package/dist/loop/runner.js +47 -31
  42. package/dist/loop/scheduler.js +8 -0
  43. package/dist/prd/command.js +6 -0
  44. package/dist/retrofit/config.js +15 -0
  45. package/dist/retrofit/planners/codex.js +64 -19
  46. package/dist/review/command.js +52 -12
  47. package/dist/review/verdict.js +45 -0
  48. package/docs/MIGRATING-TO-1.0.md +33 -0
  49. package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
  50. package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
  51. package/hooks/hooks.json +19 -0
  52. package/package.json +82 -67
  53. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
  54. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
  55. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
  56. package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
  57. package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
  58. package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
  59. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
  60. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
  61. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
  62. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
  63. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
  64. package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
  65. package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
  66. package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
  67. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
  68. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
  69. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
@@ -0,0 +1,20 @@
1
+ export class MergeQueue {
2
+ tail = Promise.resolve();
3
+ enqueue(job) {
4
+ const run = async () => {
5
+ try {
6
+ await job.rebase();
7
+ if (!await job.verify())
8
+ return { storyId: job.storyId, integrated: false, reason: 'integrated-tree verification failed' };
9
+ await job.integrate();
10
+ return { storyId: job.storyId, integrated: true };
11
+ }
12
+ catch (error) {
13
+ return { storyId: job.storyId, integrated: false, reason: error.message };
14
+ }
15
+ };
16
+ const result = this.tail.then(run, run);
17
+ this.tail = result.then(() => undefined);
18
+ return result;
19
+ }
20
+ }
@@ -0,0 +1,39 @@
1
+ import { readyStories } from './scheduler.js';
2
+ export async function runParallelLoop(stories, opts) {
3
+ const completed = [];
4
+ const failed = [];
5
+ const activeAreas = new Set();
6
+ const active = new Map();
7
+ let iterations = 0;
8
+ let agentIndex = 0;
9
+ const launch = (story) => {
10
+ iterations++;
11
+ if (story.area)
12
+ activeAreas.add(story.area);
13
+ const affinity = story.agent ?? (opts.agents?.length ? opts.agents[agentIndex++ % opts.agents.length] : undefined);
14
+ const promise = opts.worker(story, affinity).then(result => {
15
+ if (result.success) {
16
+ story.passes = true;
17
+ completed.push(story.id);
18
+ }
19
+ else
20
+ failed.push(story.id);
21
+ }).catch(() => { failed.push(story.id); }).finally(() => {
22
+ active.delete(story.id);
23
+ if (story.area)
24
+ activeAreas.delete(story.area);
25
+ });
26
+ active.set(story.id, promise);
27
+ };
28
+ while (iterations < opts.maxIterations && !opts.paused?.()) {
29
+ const slots = Math.max(0, opts.maxConcurrency - active.size);
30
+ const ready = readyStories(stories, { activeAreas }).filter(s => !active.has(s.id) && !failed.includes(s.id)).slice(0, slots);
31
+ for (const story of ready)
32
+ launch(story);
33
+ if (active.size === 0)
34
+ break;
35
+ await Promise.race(active.values());
36
+ }
37
+ await Promise.all(active.values());
38
+ return { completed, failed, iterations, paused: opts.paused?.() ?? false };
39
+ }
package/dist/loop/prd.js CHANGED
@@ -7,20 +7,66 @@ export const StorySchema = z.object({
7
7
  priority: z.number(),
8
8
  acceptance: z.array(z.string().min(1)),
9
9
  passes: z.boolean(),
10
+ needs: z.array(z.string().min(1)).optional(),
11
+ area: z.string().min(1).optional(),
12
+ agent: z.enum(['claude', 'codex', 'gemini']).optional(),
10
13
  });
11
14
  const PrdSchema = z.array(StorySchema);
12
15
  export function loadPrd(file) {
13
- return PrdSchema.parse(parse(readFileSync(file, 'utf8')));
16
+ const stories = PrdSchema.parse(parse(readFileSync(file, 'utf8')));
17
+ const issues = validateDependencies(stories);
18
+ if (issues.length)
19
+ throw new Error(`Invalid PRD dependency graph:\n${issues.join('\n')}`);
20
+ return stories;
14
21
  }
15
22
  export function savePrd(file, stories) {
16
23
  writeFileSync(file, stringify(stories));
17
24
  }
18
25
  export function selectNextStory(stories) {
19
- const open = stories.filter(s => !s.passes);
26
+ const passed = new Set(stories.filter(s => s.passes).map(s => s.id));
27
+ const open = stories.filter(s => !s.passes && (s.needs ?? []).every(id => passed.has(id)));
20
28
  if (open.length === 0)
21
29
  return null;
22
30
  return open.reduce((best, s) => (s.priority < best.priority ? s : best));
23
31
  }
32
+ export function validateDependencies(stories) {
33
+ const issues = [];
34
+ const counts = new Map();
35
+ for (const story of stories)
36
+ counts.set(story.id, (counts.get(story.id) ?? 0) + 1);
37
+ for (const [id, count] of counts)
38
+ if (count > 1)
39
+ issues.push(`duplicate story id: ${id}`);
40
+ const ids = new Set(stories.map(s => s.id));
41
+ for (const story of stories) {
42
+ for (const need of story.needs ?? []) {
43
+ if (need === story.id)
44
+ issues.push(`story ${story.id} depends on itself`);
45
+ else if (!ids.has(need))
46
+ issues.push(`story ${story.id} has unknown dependency ${need}`);
47
+ }
48
+ }
49
+ const byId = new Map(stories.map(s => [s.id, s]));
50
+ const visiting = new Set();
51
+ const visited = new Set();
52
+ const walk = (id, path) => {
53
+ if (visiting.has(id)) {
54
+ issues.push(`dependency cycle: ${[...path, id].join(' -> ')}`);
55
+ return;
56
+ }
57
+ if (visited.has(id))
58
+ return;
59
+ visiting.add(id);
60
+ for (const need of byId.get(id)?.needs ?? [])
61
+ if (byId.has(need) && need !== id)
62
+ walk(need, [...path, id]);
63
+ visiting.delete(id);
64
+ visited.add(id);
65
+ };
66
+ for (const id of byId.keys())
67
+ walk(id, []);
68
+ return [...new Set(issues)];
69
+ }
24
70
  export function allPass(stories) {
25
71
  return stories.length > 0 && stories.every(s => s.passes);
26
72
  }
@@ -9,6 +9,8 @@ import { commandVerifier, retryingVerifier } from './verify.js';
9
9
  import { readStatus, makeReporter, fmtDuration } from './reporter.js';
10
10
  import { acquireLock, releaseLock } from './lock.js';
11
11
  import { maybeAutoUpgrade } from '../update/upgrade.js';
12
+ import { resolveCommitIdentity } from './identity.js';
13
+ import { runAudit } from '../audit/command.js';
12
14
  export const DEFAULT_IDLE_MINUTES = 20;
13
15
  const STALE_MINUTES = 20; // a running status older than this likely means the loop died
14
16
  export function relativeTime(fromIso, now) {
@@ -66,6 +68,10 @@ export function resolveIdleMs(flagMinutes, configMinutes) {
66
68
  return minutes > 0 ? minutes * 60_000 : 0;
67
69
  }
68
70
  export function runLoopCommand(targetDir, opts) {
71
+ if ((opts.parallel ?? 1) > 1) {
72
+ console.error('Parallel CLI workers are not enabled yet. The dependency-aware dispatcher and merge queue are available as APIs; use --parallel=1 for the synchronous provider runner.');
73
+ return 2;
74
+ }
69
75
  const config = loadConfig(targetDir);
70
76
  if (!config?.loop.enabled) {
71
77
  console.error('Loop is disabled. Enable it with: yoke loop on');
@@ -85,12 +91,41 @@ export function runLoopCommand(targetDir, opts) {
85
91
  }
86
92
  verify = retryingVerifier(commandVerifier(command), config.verify?.retries ?? 1);
87
93
  }
94
+ // Optional performance budget gate: same contract as verify (exit 0 = within
95
+ // budget), same flake tolerance (benchmarks are noisy).
96
+ let perf = opts.perf;
97
+ if (!perf && config.perf?.command) {
98
+ perf = retryingVerifier(commandVerifier(config.perf.command), config.perf.retries ?? 1);
99
+ }
88
100
  // Opt-in self-update, loop START only — this run keeps executing the version
89
101
  // it started with; a fetched upgrade applies from the next invocation.
90
102
  maybeAutoUpgrade(config.update?.auto);
91
103
  const available = opts.isAvailable ?? isAgentAvailable;
92
104
  const runnerAgent = opts.agent ?? config.agents[0] ?? 'claude';
105
+ const git = opts.git ?? realGitOps;
106
+ let commitIdentity = opts.commitIdentity;
107
+ if (!commitIdentity && !opts.git) {
108
+ try {
109
+ commitIdentity = resolveCommitIdentity(targetDir, config.commit);
110
+ }
111
+ catch (error) {
112
+ console.error(error.message);
113
+ return 2;
114
+ }
115
+ }
116
+ let audit = opts.audit;
117
+ if (!audit && config.audit?.enabled) {
118
+ audit = (dir) => {
119
+ const result = runAudit(dir, { command: config.audit?.command, suppressions: config.audit?.suppressions });
120
+ return { passed: result.code === 0, summary: result.error ?? (result.findings.map(f => `${f.ruleId} ${f.file}${f.line ? `:${f.line}` : ''}`).join(', ') || 'audit passed') };
121
+ };
122
+ }
123
+ if (commitIdentity) {
124
+ const announce = opts.json ? console.error : console.log;
125
+ announce(`Commits: ${commitIdentity.authorName} <${commitIdentity.authorEmail}> · co-authors: ${commitIdentity.allowCoAuthors ? 'allowed' : 'disabled'}`);
126
+ }
93
127
  const idleMs = resolveIdleMs(opts.timeoutMinutes, config.loop.timeoutMinutes);
128
+ const permissions = opts.permissions ?? config.runner?.permissions ?? 'safe';
94
129
  let runner = opts.runner;
95
130
  if (!runner) {
96
131
  if (!available(runnerAgent)) {
@@ -99,16 +134,34 @@ export function runLoopCommand(targetDir, opts) {
99
134
  }
100
135
  // Token reporting is part of the machine interface: in --json mode a claude
101
136
  // runner switches to stream-json so cumulative usage rides on every status.
102
- runner = makeRunner(runnerAgent, idleMs, { tokenReport: opts.json === true, onAmbiguity: opts.onAmbiguity ?? config.loop.onAmbiguity });
137
+ runner = makeRunner(runnerAgent, idleMs, {
138
+ tokenReport: opts.json === true,
139
+ onAmbiguity: opts.onAmbiguity ?? config.loop.onAmbiguity,
140
+ perfCommand: config.perf?.command,
141
+ permissions,
142
+ });
143
+ const announce = opts.json ? console.error : console.log;
144
+ announce(`Runner: ${runnerAgent} · permissions: ${permissions} · cwd: ${targetDir}`);
103
145
  }
104
146
  let review = opts.reviewRunner;
105
147
  if (!review && (opts.review || opts.reviewer)) {
106
- const reviewerAgent = opts.reviewer ?? runnerAgent;
107
- if (!available(reviewerAgent)) {
108
- console.error(`Reviewer agent CLI "${reviewerAgent}" was not found on PATH. Install it, or pick another with --reviewer=<claude|codex|gemini>.`);
148
+ const reviewerAgent = opts.reviewer ?? ['codex', 'gemini', 'claude'].find(agent => agent !== runnerAgent && available(agent));
149
+ if (!reviewerAgent) {
150
+ if (!opts.allowSelfReview) {
151
+ console.error('No independent reviewer CLI is available. Install or select a second agent, or pass --allow-self-review explicitly.');
152
+ return 2;
153
+ }
154
+ }
155
+ const resolvedReviewer = reviewerAgent ?? runnerAgent;
156
+ if (resolvedReviewer === runnerAgent && !opts.allowSelfReview) {
157
+ console.error(`Reviewer "${resolvedReviewer}" is also the implementer. Pick another agent or pass --allow-self-review explicitly.`);
158
+ return 2;
159
+ }
160
+ if (!available(resolvedReviewer)) {
161
+ console.error(`Reviewer agent CLI "${resolvedReviewer}" was not found on PATH. Install it, or pick another with --reviewer=<claude|codex|gemini>.`);
109
162
  return 2;
110
163
  }
111
- review = makeReviewRunner(reviewerAgent, idleMs);
164
+ review = makeReviewRunner(resolvedReviewer, idleMs);
112
165
  }
113
166
  const lock = acquireLock(targetDir);
114
167
  if (!lock.acquired) {
@@ -123,10 +176,13 @@ export function runLoopCommand(targetDir, opts) {
123
176
  prdPath: path,
124
177
  targetDir,
125
178
  runner,
126
- git: opts.git ?? realGitOps,
179
+ git,
180
+ commitIdentity,
127
181
  verify,
182
+ perf,
183
+ audit,
128
184
  maxIterations: opts.maxIterations,
129
- isolate: opts.isolate ?? false,
185
+ isolate: (opts.parallel ?? 1) > 1 ? true : (opts.isolate ?? false),
130
186
  review,
131
187
  reporter: opts.reporter ?? makeReporter(targetDir, { json: opts.json }),
132
188
  });
@@ -1,12 +1,15 @@
1
1
  import { execFileSync, execSync } from 'node:child_process';
2
- import { existsSync } from 'node:fs';
2
+ import { existsSync, mkdirSync, rmSync } from 'node:fs';
3
3
  import { join } from 'node:path';
4
4
  import { fileURLToPath } from 'node:url';
5
5
  import { loadContext, formatForPrompt, contextDir } from '../context/context.js';
6
+ import { buildProviderInvocation } from '../agents/providers.js';
7
+ import { parseProviderTelemetry } from '../agents/telemetry.js';
8
+ import { formatReviewContract, readReviewVerdict, reviewVerdictPath } from '../review/verdict.js';
6
9
  export function contextBlockFor(targetDir) {
7
10
  return formatForPrompt(loadContext(contextDir(targetDir)));
8
11
  }
9
- export function buildClaudePrompt(story, context, onAmbiguity = 'resolve') {
12
+ export function buildClaudePrompt(story, context, onAmbiguity = 'resolve', perfCommand) {
10
13
  const criteria = story.acceptance.map(a => `- ${a}`).join('\n');
11
14
  const lines = [
12
15
  'You are an autonomous coding agent running inside the Yoke loop.',
@@ -16,10 +19,14 @@ export function buildClaudePrompt(story, context, onAmbiguity = 'resolve') {
16
19
  lines.push('', context);
17
20
  lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria (Definition of Done):', criteria, '', "When done, ensure the project's full test suite passes.", 'Do NOT commit — the loop commits on your behalf after verifying.', '', 'Working rules:', '- Add nothing beyond what the story requires: no extra features, abstractions, comments, or defensive code for cases that cannot happen.', '- Do not create summary, plan, or analysis documents — only files the story itself needs.', '- If a check fails, fix the root cause; never bypass it (e.g. --no-verify) or pass by weakening tests.', '- Report the outcome faithfully: if a criterion is unmet or tests fail, say so plainly instead of claiming success.', '- Never ask questions or wait for input — you run unattended and nobody can answer.', onAmbiguity === 'abort'
18
21
  ? '- If an acceptance criterion is genuinely undecidable, do NOT guess: write the open question(s) to .yoke/ambiguity.md, change nothing else, and stop.'
19
- : '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.', '- Keep your final message to a few short sentences: what changed and what you verified.');
22
+ : '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.');
23
+ if (perfCommand) {
24
+ lines.push(`- This project enforces a performance budget: \`${perfCommand}\` must exit 0 or the story is blocked. Keep hot paths efficient, and never simplify away an existing optimization without re-running that benchmark.`);
25
+ }
26
+ lines.push('- Keep your final message to a few short sentences: what changed and what you verified.');
20
27
  return lines.join('\n');
21
28
  }
22
- export function buildReviewPrompt(story, context) {
29
+ export function buildReviewPrompt(story, context, verdictPath) {
23
30
  const criteria = story.acceptance.map(a => `- ${a}`).join('\n');
24
31
  const lines = [
25
32
  'You are an independent reviewer inside the Yoke loop. You did NOT implement this change.',
@@ -27,10 +34,12 @@ export function buildReviewPrompt(story, context) {
27
34
  ];
28
35
  if (context)
29
36
  lines.push('', context);
30
- lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve by exiting 0 ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), exit non-zero to reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
37
+ lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
38
+ if (verdictPath)
39
+ lines.push('', formatReviewContract(verdictPath));
31
40
  return lines.join('\n');
32
41
  }
33
- export function buildStandaloneReviewPrompt(scope, focus) {
42
+ export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
34
43
  const lines = [
35
44
  'You are an independent reviewer. You did NOT write this change.',
36
45
  `Review ${scope}. Run git yourself to see the diff (e.g. \`git diff\`, or \`git diff <base>..HEAD\`).`,
@@ -39,6 +48,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
39
48
  if (focus)
40
49
  lines.push(`Pay particular attention to: ${focus}.`);
41
50
  lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
51
+ if (verdictPath)
52
+ lines.push('', formatReviewContract(verdictPath));
42
53
  return lines.join('\n');
43
54
  }
44
55
  // Headless agents must run non-interactively: with plain `-p` the CLI denies
@@ -47,16 +58,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
47
58
  // and falsely marks the story done. Granting autonomous permissions makes the
48
59
  // implementer actually able to write files and run the verify command.
49
60
  // (The loop is opt-in and scoped to the target project dir.)
50
- const AGENT_SPECS = {
51
- claude: { command: 'claude', baseArgs: ['-p', '--dangerously-skip-permissions'] },
52
- codex: { command: 'codex', baseArgs: ['exec', '--dangerously-bypass-approvals-and-sandbox'] },
53
- // gemini: no `-p` — current Gemini CLI (0.33+) requires a value after -p, and
54
- // piped (non-TTY) stdin already selects headless mode on its own.
55
- gemini: { command: 'gemini', baseArgs: ['--yolo'] },
56
- };
57
- export function agentInvocation(agent, prompt, cwd) {
58
- const spec = AGENT_SPECS[agent];
59
- return { command: spec.command, args: spec.baseArgs, input: prompt, cwd };
61
+ export function agentInvocation(agent, prompt, cwd, permissions = 'safe') {
62
+ return buildProviderInvocation(agent, prompt, cwd, permissions);
60
63
  }
61
64
  export function claudeInvocation(prompt, cwd) {
62
65
  return agentInvocation('claude', prompt, cwd);
@@ -65,7 +68,7 @@ export function claudeInvocation(prompt, cwd) {
65
68
  // (--verbose is required by the CLI for stream-json in -p mode). Prompt still via stdin.
66
69
  // Derived from the base spec so the headless permission-bypass flag rides along.
67
70
  export function claudeStreamJsonInvocation(prompt, cwd) {
68
- return { command: 'claude', args: [...AGENT_SPECS.claude.baseArgs, '--output-format', 'stream-json', '--verbose'], input: prompt, cwd };
71
+ return buildProviderInvocation('claude', prompt, cwd, 'safe');
69
72
  }
70
73
  // Pick the runner invocation. Claude ALWAYS runs in stream-json mode: plain `-p`
71
74
  // prints nothing until the run finishes, so the idle watchdog saw a healthy
@@ -73,10 +76,8 @@ export function claudeStreamJsonInvocation(prompt, cwd) {
73
76
  // the user saw dead air the whole time. stream-json emits per-message output,
74
77
  // which doubles as liveness. Token usage rides along for free. Other agents
75
78
  // keep their plain invocation (no machine-readable stream to gain).
76
- export function runnerInvocation(agent, prompt, cwd, _tokenReport = false) {
77
- if (agent === 'claude')
78
- return claudeStreamJsonInvocation(prompt, cwd);
79
- return agentInvocation(agent, prompt, cwd);
79
+ export function runnerInvocation(agent, prompt, cwd, _tokenReport = false, permissions = 'safe') {
80
+ return buildProviderInvocation(agent, prompt, cwd, permissions);
80
81
  }
81
82
  // Parse claude stream-json output into cumulative token usage. Defensive by design:
82
83
  // non-JSON lines and unknown message shapes are ignored. The final "result" message
@@ -226,20 +227,21 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
226
227
  // Claude always streams (see runnerInvocation) — capture the stream so tokens are
227
228
  // always reported; other agents keep inherit stdio. opts.tokenReport is now
228
229
  // redundant for claude and meaningless elsewhere; kept for caller compatibility.
229
- const captureTokens = agent === 'claude';
230
+ const captureTokens = true;
230
231
  return (ctx) => {
231
- const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity), ctx.targetDir, captureTokens);
232
+ const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens, opts.permissions ?? 'safe');
232
233
  const inv = buildWatchdogInvocation(base, idleTimeoutMs);
233
234
  if (captureTokens) {
234
235
  const capture = opts.execCapture ?? runCliCapture;
235
236
  try {
236
237
  const out = capture(inv);
237
- return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: parseClaudeStreamUsage(out.split(/\r?\n/)) };
238
+ const telemetry = parseProviderTelemetry(agent, out.split(/\r?\n/));
239
+ return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: telemetry.tokens };
238
240
  }
239
241
  catch (e) {
240
242
  // Salvage usage from whatever the agent streamed before dying — those tokens were spent.
241
243
  const partial = e.stdout;
242
- const tokens = partial == null ? undefined : parseClaudeStreamUsage(String(partial).split(/\r?\n/));
244
+ const tokens = partial == null ? undefined : parseProviderTelemetry(agent, String(partial).split(/\r?\n/)).tokens;
243
245
  return { success: false, summary: `${agent} failed on ${ctx.story.id}: ${e.message}`, tokens };
244
246
  }
245
247
  }
@@ -256,21 +258,35 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
256
258
  };
257
259
  }
258
260
  export const claudeRunner = makeRunner('claude');
259
- export function makeReviewRunner(agent, idleTimeoutMs = 0) {
261
+ export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runCli) {
260
262
  return (ctx) => {
261
- const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir)), ctx.targetDir);
263
+ const verdictPath = reviewVerdictPath(ctx.targetDir);
264
+ mkdirSync(join(ctx.targetDir, '.yoke'), { recursive: true });
265
+ rmSync(verdictPath, { force: true });
266
+ const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir), verdictPath), ctx.targetDir, 'safe');
262
267
  const inv = buildWatchdogInvocation(base, idleTimeoutMs);
268
+ let processFailure;
269
+ try {
270
+ exec(inv);
271
+ }
272
+ catch (e) {
273
+ processFailure = e.message;
274
+ }
263
275
  try {
264
- runCli(inv);
265
- return { success: true, summary: `${agent} approved ${ctx.story.id}` };
276
+ const verdict = readReviewVerdict(verdictPath);
277
+ if (processFailure)
278
+ return { success: false, summary: `review process failed: ${processFailure}; verdict: ${verdict.summary}` };
279
+ return verdict.approved
280
+ ? { success: true, summary: `${agent} approved ${ctx.story.id}: ${verdict.summary}` }
281
+ : { success: false, summary: `${agent} rejected ${ctx.story.id}: ${verdict.summary}` };
266
282
  }
267
283
  catch (e) {
268
- return { success: false, summary: `${agent} rejected ${ctx.story.id}: ${e.message}` };
284
+ return { success: false, summary: `${processFailure ? `review process failed: ${processFailure}; ` : ''}${e.message}` };
269
285
  }
270
286
  };
271
287
  }
272
288
  // Probe whether the agent's CLI is on PATH (so the loop can refuse upfront with a
273
289
  // clear message instead of failing mid-run with spawn ENOENT). Never throws.
274
290
  export function isAgentAvailable(agent) {
275
- return probeVersion(AGENT_SPECS[agent].command);
291
+ return probeVersion(agent);
276
292
  }
@@ -0,0 +1,8 @@
1
+ export function readyStories(stories, opts = {}) {
2
+ const passed = new Set(stories.filter(story => story.passes).map(story => story.id));
3
+ return stories
4
+ .filter(story => !story.passes)
5
+ .filter(story => (story.needs ?? []).every(id => passed.has(id)))
6
+ .filter(story => !story.area || !opts.activeAreas?.has(story.area))
7
+ .sort((a, b) => a.priority - b.priority || Number(b.agent === opts.agent) - Number(a.agent === opts.agent) || a.id.localeCompare(b.id));
8
+ }
@@ -9,6 +9,9 @@ export const PRD_TEMPLATE = `# Yoke PRD — the loop picks the lowest-priority o
9
9
  # - id: STORY-1
10
10
  # title: scaffold the project with a runnable test suite
11
11
  # priority: 1
12
+ # needs: [] # optional story IDs that must pass first
13
+ # area: foundation # optional collision domain for parallel runs
14
+ # agent: codex # optional claude|codex|gemini affinity
12
15
  # acceptance:
13
16
  # - "the verify command exits 0"
14
17
  # - "a placeholder test exists and passes"
@@ -26,6 +29,9 @@ export function buildPrdDraftPrompt(idea) {
26
29
  '- id: STORY-1, STORY-2, ... (unique)',
27
30
  '- title: one imperative sentence',
28
31
  '- priority: dense integers from 1 (lower = built first)',
32
+ '- needs: optional list of story IDs that must pass first; the graph must be acyclic',
33
+ '- area: optional collision domain for safe parallel scheduling',
34
+ '- agent: optional claude|codex|gemini affinity',
29
35
  '- acceptance: 2-5 testable, behavioral criteria (observable outcomes, never implementation steps)',
30
36
  '- passes: false',
31
37
  '',
@@ -16,7 +16,22 @@ export const YokeConfigSchema = z.object({
16
16
  // or 'abort' (agent stops the story via .yoke/ambiguity.md for a human decision).
17
17
  onAmbiguity: z.enum(['resolve', 'abort']).optional(),
18
18
  }),
19
+ runner: z.object({ permissions: z.enum(['safe', 'unsafe', 'read-only']) }).optional(),
20
+ commit: z.object({
21
+ authorName: z.string().min(1).optional(),
22
+ authorEmail: z.string().email().optional(),
23
+ allowCoAuthors: z.boolean().optional(),
24
+ }).optional(),
25
+ audit: z.object({
26
+ enabled: z.boolean(),
27
+ command: z.string().min(1).optional(),
28
+ suppressionsVersion: z.literal(1).optional(),
29
+ suppressions: z.array(z.object({ ruleId: z.string().min(1), file: z.string().min(1).optional(), reason: z.string(), expires: z.string().optional() })).optional(),
30
+ }).optional(),
19
31
  verify: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
32
+ // Optional performance budget gate: a benchmark command that must exit 0 for a
33
+ // story to land (runs after verify). Benchmarks are noisy → retried like verify.
34
+ perf: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
20
35
  codeGraph: CodeGraphSchema.optional(),
21
36
  smoke: SmokeSchema.optional(),
22
37
  // Opt-in: upgrade yoke at loop START when a newer version is cached (never mid-run).
@@ -1,5 +1,6 @@
1
1
  import { readFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
+ import { loadManifest } from '../../canon/manifest.js';
3
4
  import { mcpServers, rtkInstruction } from '../tools.js';
4
5
  function tomlMcp(codeGraph) {
5
6
  const servers = mcpServers(codeGraph);
@@ -13,24 +14,68 @@ function tomlMcp(codeGraph) {
13
14
  .join('\n');
14
15
  }
15
16
  export function planCodex(canonDir, _targetDir, codeGraph = 'graphify') {
16
- return [
17
- {
18
- kind: 'write',
19
- target: 'AGENTS.md',
20
- content: readFileSync(join(canonDir, 'AGENTS.md'), 'utf8'),
21
- reason: 'baseline instructions (Codex reads AGENTS.md natively)',
22
- },
23
- {
24
- kind: 'write',
25
- target: '.codex/config.toml',
26
- content: `# Yoke: MCP servers for Codex. Merge into ~/.codex/config.toml.\n\n${tomlMcp(codeGraph)}`,
27
- reason: 'MCP servers (code-graph + playwright)',
28
- },
29
- {
30
- kind: 'write',
31
- target: 'RTK.md',
32
- content: rtkInstruction() + '\n',
33
- reason: 'rtk instruction (Codex has no rewrite hook)',
34
- },
17
+ const manifest = loadManifest(join(canonDir, 'manifest.yaml'));
18
+ const baseline = readFileSync(join(canonDir, 'AGENTS.md'), 'utf8');
19
+ const actions = manifest.skills.map(skill => ({
20
+ kind: 'write',
21
+ target: `.agents/skills/${skill.id}/SKILL.md`,
22
+ content: readFileSync(join(canonDir, skill.path, 'SKILL.md'), 'utf8'),
23
+ reason: `skill: ${skill.id}`,
24
+ }));
25
+ const roles = [
26
+ ['implementer', 'Implementation specialist for one scoped story.', 'workspace-write', 'Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.'],
27
+ ['reviewer', 'Read-only reviewer for correctness and acceptance criteria.', 'read-only', 'Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.'],
28
+ ['security', 'Read-only security reviewer for changed code.', 'read-only', 'Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.'],
29
+ ['docs', 'Documentation specialist for release and API consistency.', 'workspace-write', 'Update only documentation required by the assigned change. Verify commands and version references against the repository.'],
35
30
  ];
31
+ actions.push({
32
+ kind: 'write',
33
+ target: 'AGENTS.md',
34
+ content: `${baseline.trimEnd()}\n\n@RTK.md\n`,
35
+ reason: 'baseline instructions (Codex reads AGENTS.md natively)',
36
+ }, {
37
+ kind: 'write',
38
+ target: '.codex/config.toml',
39
+ content: `# Yoke project configuration. Codex loads this in trusted repositories.\n\n[features]\nhooks = true\n\n${tomlMcp(codeGraph)}`,
40
+ reason: 'MCP servers (code-graph + playwright)',
41
+ }, {
42
+ kind: 'write',
43
+ target: '.codex/hooks.json',
44
+ merge: true,
45
+ content: JSON.stringify({
46
+ description: 'Yoke command compression for Codex',
47
+ hooks: {
48
+ PreToolUse: [{
49
+ matcher: '^Bash$',
50
+ hooks: [{
51
+ type: 'command',
52
+ command: 'node "$(git rev-parse --show-toplevel)/.codex/hooks/rtk.mjs"',
53
+ commandWindows: 'powershell -NoProfile -ExecutionPolicy Bypass -Command "$root = git rev-parse --show-toplevel; node (Join-Path $root \'.codex/hooks/rtk.mjs\')"',
54
+ timeout: 5,
55
+ statusMessage: 'Compressing command output with RTK',
56
+ }],
57
+ }],
58
+ },
59
+ }, null, 2) + '\n',
60
+ reason: 'rtk PreToolUse hook adapter',
61
+ }, {
62
+ kind: 'write',
63
+ target: '.codex/hooks/rtk.mjs',
64
+ content: readFileSync(join(canonDir, 'tools', 'codex-rtk-hook.mjs'), 'utf8'),
65
+ reason: 'rtk Codex hook adapter',
66
+ }, {
67
+ kind: 'write',
68
+ target: 'RTK.md',
69
+ content: rtkInstruction() + '\n',
70
+ reason: 'rtk instruction (Codex has no rewrite hook)',
71
+ });
72
+ for (const [name, description, sandbox, instructions] of roles) {
73
+ actions.push({
74
+ kind: 'write',
75
+ target: `.codex/agents/${name}.toml`,
76
+ content: `name = "${name}"\ndescription = "${description}"\nsandbox_mode = "${sandbox}"\ndeveloper_instructions = """\n${instructions}\n"""\n`,
77
+ reason: `Codex role agent: ${name}`,
78
+ });
79
+ }
80
+ return actions;
36
81
  }