@hecer/yoke 0.9.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/.claude-plugin/marketplace.json +18 -0
  2. package/.claude-plugin/plugin.json +13 -0
  3. package/.codex-plugin/plugin.json +7 -0
  4. package/CHANGELOG.md +192 -149
  5. package/README.md +101 -51
  6. package/TODOS.md +8 -0
  7. package/agents/docs.toml +6 -0
  8. package/agents/implementer.toml +6 -0
  9. package/agents/reviewer.toml +6 -0
  10. package/agents/security.toml +6 -0
  11. package/bench/README.md +45 -42
  12. package/bench/RESULTS.md +46 -36
  13. package/bench/result-schema.mjs +12 -0
  14. package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
  15. package/bench/results/codex-unavailable-1785175418318.json +15 -0
  16. package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
  17. package/bench/run-matrix.mjs +26 -0
  18. package/bench/run.mjs +127 -115
  19. package/canon/AGENTS.md +2 -0
  20. package/canon/loop/loop-spec.md +4 -2
  21. package/canon/loop/prd.schema.md +5 -0
  22. package/canon/manifest.yaml +2 -1
  23. package/canon/skills/authoring-prd/SKILL.md +10 -3
  24. package/canon/skills/ship/SKILL.md +2 -7
  25. package/canon/skills/workflow/SKILL.md +4 -0
  26. package/canon/skills/yoke-retrofit/SKILL.md +18 -11
  27. package/canon/skills/yoke-workflow/SKILL.md +20 -0
  28. package/canon/tools/codex-rtk-hook.mjs +36 -0
  29. package/dist/agents/host.js +26 -0
  30. package/dist/agents/providers.js +23 -0
  31. package/dist/agents/telemetry.js +30 -0
  32. package/dist/agents/types.js +1 -0
  33. package/dist/audit/changes.js +6 -0
  34. package/dist/audit/command.js +64 -0
  35. package/dist/audit/dependencies.js +21 -0
  36. package/dist/audit/secrets.js +16 -0
  37. package/dist/audit/types.js +1 -0
  38. package/dist/cli.js +189 -6
  39. package/dist/context/context.js +15 -2
  40. package/dist/loop/claims.js +57 -0
  41. package/dist/loop/cleanup.js +98 -27
  42. package/dist/loop/decision.js +517 -0
  43. package/dist/loop/git.js +31 -2
  44. package/dist/loop/identity.js +27 -0
  45. package/dist/loop/lock.js +104 -13
  46. package/dist/loop/loop.js +49 -2
  47. package/dist/loop/merge-queue.js +20 -0
  48. package/dist/loop/parallel.js +39 -0
  49. package/dist/loop/prd.js +48 -2
  50. package/dist/loop/run-command.js +118 -12
  51. package/dist/loop/runner.js +48 -30
  52. package/dist/loop/scheduler.js +8 -0
  53. package/dist/prd/command.js +30 -21
  54. package/dist/retrofit/command.js +3 -2
  55. package/dist/retrofit/config.js +16 -0
  56. package/dist/retrofit/gitignore.js +8 -0
  57. package/dist/retrofit/planners/codex.js +64 -19
  58. package/dist/retrofit/report.js +1 -1
  59. package/dist/review/command.js +52 -12
  60. package/dist/review/verdict.js +45 -0
  61. package/dist/setup/command.js +82 -0
  62. package/docs/MIGRATING-TO-1.0.md +33 -0
  63. package/docs/MIGRATING-TO-1.1.md +27 -0
  64. package/docs/PUBLISHING.md +77 -41
  65. package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
  66. package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
  67. package/gemini-extension.json +6 -0
  68. package/hooks/hooks.json +19 -0
  69. package/package.json +84 -67
  70. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
  71. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
  72. package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
  73. package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
  74. package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
  75. package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
  76. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
  77. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
  78. package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
  79. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
  80. package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
  81. package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
  82. package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
  83. package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
  84. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
  85. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
  86. package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
@@ -1,8 +1,11 @@
1
1
  import { execFileSync, execSync } from 'node:child_process';
2
- import { existsSync } from 'node:fs';
2
+ import { existsSync, mkdirSync, rmSync } from 'node:fs';
3
3
  import { join } from 'node:path';
4
4
  import { fileURLToPath } from 'node:url';
5
5
  import { loadContext, formatForPrompt, contextDir } from '../context/context.js';
6
+ import { buildProviderInvocation } from '../agents/providers.js';
7
+ import { parseProviderTelemetry } from '../agents/telemetry.js';
8
+ import { formatReviewContract, readReviewVerdict, reviewVerdictPath } from '../review/verdict.js';
6
9
  export function contextBlockFor(targetDir) {
7
10
  return formatForPrompt(loadContext(contextDir(targetDir)));
8
11
  }
@@ -16,14 +19,20 @@ export function buildClaudePrompt(story, context, onAmbiguity = 'resolve', perfC
16
19
  lines.push('', context);
17
20
  lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria (Definition of Done):', criteria, '', "When done, ensure the project's full test suite passes.", 'Do NOT commit — the loop commits on your behalf after verifying.', '', 'Working rules:', '- Add nothing beyond what the story requires: no extra features, abstractions, comments, or defensive code for cases that cannot happen.', '- Do not create summary, plan, or analysis documents — only files the story itself needs.', '- If a check fails, fix the root cause; never bypass it (e.g. --no-verify) or pass by weakening tests.', '- Report the outcome faithfully: if a criterion is unmet or tests fail, say so plainly instead of claiming success.', '- Never ask questions or wait for input — you run unattended and nobody can answer.', onAmbiguity === 'abort'
18
21
  ? '- If an acceptance criterion is genuinely undecidable, do NOT guess: write the open question(s) to .yoke/ambiguity.md, change nothing else, and stop.'
19
- : '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.');
22
+ : onAmbiguity === 'critical'
23
+ ? [
24
+ '- Resolve routine ambiguity yourself using the plan, acceptance criteria, existing code, and established project conventions.',
25
+ '- Stop only for a high-impact decision involving public architecture, security or privacy posture, destructive data migration or data loss, material external cost, legal/compliance exposure, or another irreversible choice.',
26
+ '- For such a critical decision, change nothing else. Write .yoke/decision-request.yaml with exactly: version: 1, storyId, question, reason, 2-4 options ({id, label, optional tradeoff}), and recommended (an option id). Then stop.',
27
+ ].join('\n')
28
+ : '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.');
20
29
  if (perfCommand) {
21
30
  lines.push(`- This project enforces a performance budget: \`${perfCommand}\` must exit 0 or the story is blocked. Keep hot paths efficient, and never simplify away an existing optimization without re-running that benchmark.`);
22
31
  }
23
32
  lines.push('- Keep your final message to a few short sentences: what changed and what you verified.');
24
33
  return lines.join('\n');
25
34
  }
26
- export function buildReviewPrompt(story, context) {
35
+ export function buildReviewPrompt(story, context, verdictPath) {
27
36
  const criteria = story.acceptance.map(a => `- ${a}`).join('\n');
28
37
  const lines = [
29
38
  'You are an independent reviewer inside the Yoke loop. You did NOT implement this change.',
@@ -31,10 +40,12 @@ export function buildReviewPrompt(story, context) {
31
40
  ];
32
41
  if (context)
33
42
  lines.push('', context);
34
- lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve by exiting 0 ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), exit non-zero to reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
43
+ lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
44
+ if (verdictPath)
45
+ lines.push('', formatReviewContract(verdictPath));
35
46
  return lines.join('\n');
36
47
  }
37
- export function buildStandaloneReviewPrompt(scope, focus) {
48
+ export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
38
49
  const lines = [
39
50
  'You are an independent reviewer. You did NOT write this change.',
40
51
  `Review ${scope}. Run git yourself to see the diff (e.g. \`git diff\`, or \`git diff <base>..HEAD\`).`,
@@ -43,6 +54,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
43
54
  if (focus)
44
55
  lines.push(`Pay particular attention to: ${focus}.`);
45
56
  lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
57
+ if (verdictPath)
58
+ lines.push('', formatReviewContract(verdictPath));
46
59
  return lines.join('\n');
47
60
  }
48
61
  // Headless agents must run non-interactively: with plain `-p` the CLI denies
@@ -51,16 +64,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
51
64
  // and falsely marks the story done. Granting autonomous permissions makes the
52
65
  // implementer actually able to write files and run the verify command.
53
66
  // (The loop is opt-in and scoped to the target project dir.)
54
- const AGENT_SPECS = {
55
- claude: { command: 'claude', baseArgs: ['-p', '--dangerously-skip-permissions'] },
56
- codex: { command: 'codex', baseArgs: ['exec', '--dangerously-bypass-approvals-and-sandbox'] },
57
- // gemini: no `-p` — current Gemini CLI (0.33+) requires a value after -p, and
58
- // piped (non-TTY) stdin already selects headless mode on its own.
59
- gemini: { command: 'gemini', baseArgs: ['--yolo'] },
60
- };
61
- export function agentInvocation(agent, prompt, cwd) {
62
- const spec = AGENT_SPECS[agent];
63
- return { command: spec.command, args: spec.baseArgs, input: prompt, cwd };
67
+ export function agentInvocation(agent, prompt, cwd, permissions = 'safe') {
68
+ return buildProviderInvocation(agent, prompt, cwd, permissions);
64
69
  }
65
70
  export function claudeInvocation(prompt, cwd) {
66
71
  return agentInvocation('claude', prompt, cwd);
@@ -69,7 +74,7 @@ export function claudeInvocation(prompt, cwd) {
69
74
  // (--verbose is required by the CLI for stream-json in -p mode). Prompt still via stdin.
70
75
  // Derived from the base spec so the headless permission-bypass flag rides along.
71
76
  export function claudeStreamJsonInvocation(prompt, cwd) {
72
- return { command: 'claude', args: [...AGENT_SPECS.claude.baseArgs, '--output-format', 'stream-json', '--verbose'], input: prompt, cwd };
77
+ return buildProviderInvocation('claude', prompt, cwd, 'safe');
73
78
  }
74
79
  // Pick the runner invocation. Claude ALWAYS runs in stream-json mode: plain `-p`
75
80
  // prints nothing until the run finishes, so the idle watchdog saw a healthy
@@ -77,10 +82,8 @@ export function claudeStreamJsonInvocation(prompt, cwd) {
77
82
  // the user saw dead air the whole time. stream-json emits per-message output,
78
83
  // which doubles as liveness. Token usage rides along for free. Other agents
79
84
  // keep their plain invocation (no machine-readable stream to gain).
80
- export function runnerInvocation(agent, prompt, cwd, _tokenReport = false) {
81
- if (agent === 'claude')
82
- return claudeStreamJsonInvocation(prompt, cwd);
83
- return agentInvocation(agent, prompt, cwd);
85
+ export function runnerInvocation(agent, prompt, cwd, _tokenReport = false, permissions = 'safe') {
86
+ return buildProviderInvocation(agent, prompt, cwd, permissions);
84
87
  }
85
88
  // Parse claude stream-json output into cumulative token usage. Defensive by design:
86
89
  // non-JSON lines and unknown message shapes are ignored. The final "result" message
@@ -230,20 +233,21 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
230
233
  // Claude always streams (see runnerInvocation) — capture the stream so tokens are
231
234
  // always reported; other agents keep inherit stdio. opts.tokenReport is now
232
235
  // redundant for claude and meaningless elsewhere; kept for caller compatibility.
233
- const captureTokens = agent === 'claude';
236
+ const captureTokens = true;
234
237
  return (ctx) => {
235
- const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens);
238
+ const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens, opts.permissions ?? 'safe');
236
239
  const inv = buildWatchdogInvocation(base, idleTimeoutMs);
237
240
  if (captureTokens) {
238
241
  const capture = opts.execCapture ?? runCliCapture;
239
242
  try {
240
243
  const out = capture(inv);
241
- return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: parseClaudeStreamUsage(out.split(/\r?\n/)) };
244
+ const telemetry = parseProviderTelemetry(agent, out.split(/\r?\n/));
245
+ return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: telemetry.tokens };
242
246
  }
243
247
  catch (e) {
244
248
  // Salvage usage from whatever the agent streamed before dying — those tokens were spent.
245
249
  const partial = e.stdout;
246
- const tokens = partial == null ? undefined : parseClaudeStreamUsage(String(partial).split(/\r?\n/));
250
+ const tokens = partial == null ? undefined : parseProviderTelemetry(agent, String(partial).split(/\r?\n/)).tokens;
247
251
  return { success: false, summary: `${agent} failed on ${ctx.story.id}: ${e.message}`, tokens };
248
252
  }
249
253
  }
@@ -260,21 +264,35 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
260
264
  };
261
265
  }
262
266
  export const claudeRunner = makeRunner('claude');
263
- export function makeReviewRunner(agent, idleTimeoutMs = 0) {
267
+ export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runCli) {
264
268
  return (ctx) => {
265
- const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir)), ctx.targetDir);
269
+ const verdictPath = reviewVerdictPath(ctx.targetDir);
270
+ mkdirSync(join(ctx.targetDir, '.yoke'), { recursive: true });
271
+ rmSync(verdictPath, { force: true });
272
+ const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir), verdictPath), ctx.targetDir, 'safe');
266
273
  const inv = buildWatchdogInvocation(base, idleTimeoutMs);
274
+ let processFailure;
267
275
  try {
268
- runCli(inv);
269
- return { success: true, summary: `${agent} approved ${ctx.story.id}` };
276
+ exec(inv);
270
277
  }
271
278
  catch (e) {
272
- return { success: false, summary: `${agent} rejected ${ctx.story.id}: ${e.message}` };
279
+ processFailure = e.message;
280
+ }
281
+ try {
282
+ const verdict = readReviewVerdict(verdictPath);
283
+ if (processFailure)
284
+ return { success: false, summary: `review process failed: ${processFailure}; verdict: ${verdict.summary}` };
285
+ return verdict.approved
286
+ ? { success: true, summary: `${agent} approved ${ctx.story.id}: ${verdict.summary}` }
287
+ : { success: false, summary: `${agent} rejected ${ctx.story.id}: ${verdict.summary}` };
288
+ }
289
+ catch (e) {
290
+ return { success: false, summary: `${processFailure ? `review process failed: ${processFailure}; ` : ''}${e.message}` };
273
291
  }
274
292
  };
275
293
  }
276
294
  // Probe whether the agent's CLI is on PATH (so the loop can refuse upfront with a
277
295
  // clear message instead of failing mid-run with spawn ENOENT). Never throws.
278
296
  export function isAgentAvailable(agent) {
279
- return probeVersion(AGENT_SPECS[agent].command);
297
+ return probeVersion(agent);
280
298
  }
@@ -0,0 +1,8 @@
1
+ export function readyStories(stories, opts = {}) {
2
+ const passed = new Set(stories.filter(story => story.passes).map(story => story.id));
3
+ return stories
4
+ .filter(story => !story.passes)
5
+ .filter(story => (story.needs ?? []).every(id => passed.has(id)))
6
+ .filter(story => !story.area || !opts.activeAreas?.has(story.area))
7
+ .sort((a, b) => a.priority - b.priority || Number(b.agent === opts.agent) - Number(a.agent === opts.agent) || a.id.localeCompare(b.id));
8
+ }
@@ -1,41 +1,37 @@
1
- import { existsSync } from 'node:fs';
1
+ import { existsSync, readFileSync, statSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
3
  import { loadConfig } from '../retrofit/config.js';
4
4
  import { loadPrd, progress } from '../loop/prd.js';
5
5
  import { agentInvocation, buildWatchdogInvocation, runAgent, isAgentAvailable, } from '../loop/runner.js';
6
6
  import { resolveIdleMs } from '../loop/run-command.js';
7
+ import { detectHostAgent, resolveRunnerAgent } from '../agents/host.js';
7
8
  export const PRD_TEMPLATE = `# Yoke PRD — the loop picks the lowest-priority open story each iteration.
8
9
  # Story format (see canon/loop/prd.schema.md):
9
10
  # - id: STORY-1
10
11
  # title: scaffold the project with a runnable test suite
11
12
  # priority: 1
13
+ # needs: [] # optional story IDs that must pass first
14
+ # area: foundation # optional collision domain for parallel runs
15
+ # agent: codex # optional claude|codex|gemini affinity
12
16
  # acceptance:
13
17
  # - "the verify command exits 0"
14
18
  # - "a placeholder test exists and passes"
15
19
  # passes: false
16
20
  []
17
21
  `;
18
- export function buildPrdDraftPrompt(idea) {
19
- return [
22
+ export const MAX_PLANNING_BRIEF_CHARS = 20_000;
23
+ export const MAX_PLANNING_BRIEF_BYTES = MAX_PLANNING_BRIEF_CHARS * 4;
24
+ export function buildPrdDraftPrompt(idea, planningBrief) {
25
+ const lines = [
20
26
  'You are drafting a PRD for the Yoke autonomous loop.',
21
27
  '',
22
28
  `Product idea: ${idea}`,
23
- '',
24
- 'Break the idea into 5-12 small, independently shippable stories; each must fit one loop iteration.',
25
- 'Each story needs:',
26
- '- id: STORY-1, STORY-2, ... (unique)',
27
- '- title: one imperative sentence',
28
- '- priority: dense integers from 1 (lower = built first)',
29
- '- acceptance: 2-5 testable, behavioral criteria (observable outcomes, never implementation steps)',
30
- '- passes: false',
31
- '',
32
- 'If the project has no source code yet, STORY-1 must scaffold the project skeleton with a runnable',
33
- 'test suite, and its acceptance must include that the verify command (verify.command in',
34
- '.yoke/config.yaml) exits 0.',
35
- '',
36
- 'Write ONLY the file .yoke/prd.yaml as a YAML array of stories in exactly that shape.',
37
- 'Do not modify any other file. Do not commit.',
38
- ].join('\n');
29
+ ];
30
+ if (planningBrief?.trim()) {
31
+ lines.push('', '## Approved planning brief (treat these decisions as settled)', planningBrief.trim(), '', 'Do not reopen settled choices or invent alternatives that contradict this brief.');
32
+ }
33
+ lines.push('', 'Break the idea into 5-12 small, independently shippable stories; each must fit one loop iteration.', 'Each story needs:', '- id: STORY-1, STORY-2, ... (unique)', '- title: one imperative sentence', '- priority: dense integers from 1 (lower = built first)', '- needs: optional list of story IDs that must pass first; the graph must be acyclic', '- area: optional collision domain for safe parallel scheduling', '- agent: optional claude|codex|gemini affinity', '- acceptance: 2-5 testable, behavioral criteria (observable outcomes, never implementation steps)', '- passes: false', '', 'If the project has no source code yet, STORY-1 must scaffold the project skeleton with a runnable', 'test suite, and its acceptance must include that the verify command (verify.command in', '.yoke/config.yaml) exits 0.', '', 'Write ONLY the file .yoke/prd.yaml as a YAML array of stories in exactly that shape.', 'Do not modify any other file. Do not commit.');
34
+ return lines.join('\n');
39
35
  }
40
36
  export function prdFile(targetDir) {
41
37
  return join(targetDir, '.yoke', 'prd.yaml');
@@ -63,13 +59,23 @@ export function runPrdDraft(targetDir, opts) {
63
59
  }
64
60
  const available = opts.isAvailable ?? isAgentAvailable;
65
61
  const config = loadConfig(targetDir);
66
- const agent = opts.runner ?? config?.agents[0] ?? 'claude';
62
+ const agent = resolveRunnerAgent(config, opts.runner, detectHostAgent());
67
63
  if (!available(agent)) {
68
64
  console.error(`Agent CLI "${agent}" was not found on PATH. Install it, or pick another with --runner=<claude|codex|gemini>.`);
69
65
  return 2;
70
66
  }
71
67
  const idleMs = resolveIdleMs(opts.timeoutMinutes, undefined);
72
- const inv = agentInvocation(agent, buildPrdDraftPrompt(idea), targetDir);
68
+ const planPath = join(targetDir, '.yoke', 'plan.md');
69
+ if (existsSync(planPath) && statSync(planPath).size > MAX_PLANNING_BRIEF_BYTES) {
70
+ console.error(`Approved plan is too large (${statSync(planPath).size} bytes; maximum ${MAX_PLANNING_BRIEF_BYTES}). Split or condense .yoke/plan.md before drafting the PRD.`);
71
+ return 1;
72
+ }
73
+ const planningBrief = existsSync(planPath) ? readFileSync(planPath, 'utf8') : undefined;
74
+ if (planningBrief && planningBrief.length > MAX_PLANNING_BRIEF_CHARS) {
75
+ console.error(`Approved plan is too large (${planningBrief.length} characters; maximum ${MAX_PLANNING_BRIEF_CHARS}). Split or condense .yoke/plan.md before drafting the PRD.`);
76
+ return 1;
77
+ }
78
+ const inv = agentInvocation(agent, buildPrdDraftPrompt(idea, planningBrief), targetDir);
73
79
  console.log(`Drafting PRD with ${agent}...`);
74
80
  const run = opts.run ?? ((i) => runAgent(buildWatchdogInvocation(i, idleMs)));
75
81
  const result = run(inv);
@@ -117,6 +123,9 @@ export function runPrdCheck(targetDir) {
117
123
  // the schema allows [], but the loop's stop-the-line gate blocks it — fail fast here
118
124
  if (s.acceptance.length === 0)
119
125
  errors.push(`story ${s.id} has no acceptance criteria`);
126
+ if (s.acceptance.some(criterion => /\b(?:TBD|TODO|TO BE DECIDED|DECIDE LATER)\b|\?\?\?/i.test(criterion))) {
127
+ errors.push(`story ${s.id} has unresolved planning decisions in acceptance criteria`);
128
+ }
120
129
  }
121
130
  if (errors.length > 0) {
122
131
  for (const e of errors)
@@ -7,13 +7,14 @@ import { detectProject } from './detect.js';
7
7
  import { ensureGitignore } from './gitignore.js';
8
8
  import { loadConfig, saveConfig, defaultConfig } from './config.js';
9
9
  import { loadManifest } from '../canon/manifest.js';
10
+ import { detectHostAgent } from '../agents/host.js';
10
11
  export function runRetrofit(targetDir, opts) {
11
12
  const canonDir = resolveCanonDir();
12
13
  const canonVersion = loadManifest(join(canonDir, 'manifest.yaml')).version;
13
14
  const detection = detectProject(targetDir);
14
15
  const agents = opts.agents && opts.agents.length > 0
15
16
  ? opts.agents
16
- : (detection.agents.length > 0 ? detection.agents : ['claude']);
17
+ : (detection.agents.length > 0 ? detection.agents : [opts.host ?? detectHostAgent() ?? 'claude']);
17
18
  const existing = loadConfig(targetDir);
18
19
  const codeGraph = opts.codeGraph ?? existing?.codeGraph ?? 'graphify';
19
20
  const actions = planRetrofit(canonDir, targetDir, agents, codeGraph);
@@ -28,7 +29,7 @@ export function runRetrofit(targetDir, opts) {
28
29
  ...(existing ?? defaultConfig(canonVersion)),
29
30
  canonVersion,
30
31
  agents: mergedAgents,
31
- loop: { enabled: opts.loop },
32
+ loop: { ...existing?.loop, enabled: opts.loop },
32
33
  codeGraph,
33
34
  };
34
35
  saveConfig(targetDir, config);
@@ -12,10 +12,26 @@ export const YokeConfigSchema = z.object({
12
12
  loop: z.object({
13
13
  enabled: z.boolean(),
14
14
  timeoutMinutes: z.number().optional(),
15
+ decisionPolicy: z.enum(['auto', 'critical']).optional(),
15
16
  // Ambiguous acceptance criteria: 'resolve' (default — agent decides and continues)
16
17
  // or 'abort' (agent stops the story via .yoke/ambiguity.md for a human decision).
17
18
  onAmbiguity: z.enum(['resolve', 'abort']).optional(),
18
19
  }),
20
+ runner: z.object({
21
+ agent: AgentSchema.optional(),
22
+ permissions: z.enum(['safe', 'unsafe', 'read-only']).optional(),
23
+ }).optional(),
24
+ commit: z.object({
25
+ authorName: z.string().min(1).optional(),
26
+ authorEmail: z.string().email().optional(),
27
+ allowCoAuthors: z.boolean().optional(),
28
+ }).optional(),
29
+ audit: z.object({
30
+ enabled: z.boolean(),
31
+ command: z.string().min(1).optional(),
32
+ suppressionsVersion: z.literal(1).optional(),
33
+ suppressions: z.array(z.object({ ruleId: z.string().min(1), file: z.string().min(1).optional(), reason: z.string(), expires: z.string().optional() })).optional(),
34
+ }).optional(),
19
35
  verify: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
20
36
  // Optional performance budget gate: a benchmark command that must exit 0 for a
21
37
  // story to land (runs after verify). Benchmarks are noisy → retried like verify.
@@ -6,9 +6,17 @@ export const YOKE_IGNORE_LINES = [
6
6
  '.yoke/loop-status.json',
7
7
  '.yoke/loop.log',
8
8
  '.yoke/loop.lock',
9
+ '.yoke/loop.lock.takeover',
10
+ '.yoke/loop.lock.takeover.recovery',
11
+ '.yoke/loop.lock.*.tmp',
9
12
  '.yoke/loop.pause',
10
13
  '.yoke/runner.pid',
11
14
  '.yoke/ambiguity.md',
15
+ '.yoke/decision-request.yaml',
16
+ '.yoke/pending-decision.yaml',
17
+ '.yoke/decision-answering.yaml',
18
+ '.yoke/decision-resume*.yaml',
19
+ '.yoke/decision-*.yaml.*.tmp',
12
20
  '.yoke/story-durations.json',
13
21
  '.yoke/proof/',
14
22
  ];
@@ -1,5 +1,6 @@
1
1
  import { readFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
+ import { loadManifest } from '../../canon/manifest.js';
3
4
  import { mcpServers, rtkInstruction } from '../tools.js';
4
5
  function tomlMcp(codeGraph) {
5
6
  const servers = mcpServers(codeGraph);
@@ -13,24 +14,68 @@ function tomlMcp(codeGraph) {
13
14
  .join('\n');
14
15
  }
15
16
  export function planCodex(canonDir, _targetDir, codeGraph = 'graphify') {
16
- return [
17
- {
18
- kind: 'write',
19
- target: 'AGENTS.md',
20
- content: readFileSync(join(canonDir, 'AGENTS.md'), 'utf8'),
21
- reason: 'baseline instructions (Codex reads AGENTS.md natively)',
22
- },
23
- {
24
- kind: 'write',
25
- target: '.codex/config.toml',
26
- content: `# Yoke: MCP servers for Codex. Merge into ~/.codex/config.toml.\n\n${tomlMcp(codeGraph)}`,
27
- reason: 'MCP servers (code-graph + playwright)',
28
- },
29
- {
30
- kind: 'write',
31
- target: 'RTK.md',
32
- content: rtkInstruction() + '\n',
33
- reason: 'rtk instruction (Codex has no rewrite hook)',
34
- },
17
+ const manifest = loadManifest(join(canonDir, 'manifest.yaml'));
18
+ const baseline = readFileSync(join(canonDir, 'AGENTS.md'), 'utf8');
19
+ const actions = manifest.skills.map(skill => ({
20
+ kind: 'write',
21
+ target: `.agents/skills/${skill.id}/SKILL.md`,
22
+ content: readFileSync(join(canonDir, skill.path, 'SKILL.md'), 'utf8'),
23
+ reason: `skill: ${skill.id}`,
24
+ }));
25
+ const roles = [
26
+ ['implementer', 'Implementation specialist for one scoped story.', 'workspace-write', 'Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.'],
27
+ ['reviewer', 'Read-only reviewer for correctness and acceptance criteria.', 'read-only', 'Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.'],
28
+ ['security', 'Read-only security reviewer for changed code.', 'read-only', 'Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.'],
29
+ ['docs', 'Documentation specialist for release and API consistency.', 'workspace-write', 'Update only documentation required by the assigned change. Verify commands and version references against the repository.'],
35
30
  ];
31
+ actions.push({
32
+ kind: 'write',
33
+ target: 'AGENTS.md',
34
+ content: `${baseline.trimEnd()}\n\n@RTK.md\n`,
35
+ reason: 'baseline instructions (Codex reads AGENTS.md natively)',
36
+ }, {
37
+ kind: 'write',
38
+ target: '.codex/config.toml',
39
+ content: `# Yoke project configuration. Codex loads this in trusted repositories.\n\n[features]\nhooks = true\n\n${tomlMcp(codeGraph)}`,
40
+ reason: 'MCP servers (code-graph + playwright)',
41
+ }, {
42
+ kind: 'write',
43
+ target: '.codex/hooks.json',
44
+ merge: true,
45
+ content: JSON.stringify({
46
+ description: 'Yoke command compression for Codex',
47
+ hooks: {
48
+ PreToolUse: [{
49
+ matcher: '^Bash$',
50
+ hooks: [{
51
+ type: 'command',
52
+ command: 'node "$(git rev-parse --show-toplevel)/.codex/hooks/rtk.mjs"',
53
+ commandWindows: 'powershell -NoProfile -ExecutionPolicy Bypass -Command "$root = git rev-parse --show-toplevel; node (Join-Path $root \'.codex/hooks/rtk.mjs\')"',
54
+ timeout: 5,
55
+ statusMessage: 'Compressing command output with RTK',
56
+ }],
57
+ }],
58
+ },
59
+ }, null, 2) + '\n',
60
+ reason: 'rtk PreToolUse hook adapter',
61
+ }, {
62
+ kind: 'write',
63
+ target: '.codex/hooks/rtk.mjs',
64
+ content: readFileSync(join(canonDir, 'tools', 'codex-rtk-hook.mjs'), 'utf8'),
65
+ reason: 'rtk Codex hook adapter',
66
+ }, {
67
+ kind: 'write',
68
+ target: 'RTK.md',
69
+ content: rtkInstruction() + '\n',
70
+ reason: 'rtk instruction (Codex has no rewrite hook)',
71
+ });
72
+ for (const [name, description, sandbox, instructions] of roles) {
73
+ actions.push({
74
+ kind: 'write',
75
+ target: `.codex/agents/${name}.toml`,
76
+ content: `name = "${name}"\ndescription = "${description}"\nsandbox_mode = "${sandbox}"\ndeveloper_instructions = """\n${instructions}\n"""\n`,
77
+ reason: `Codex role agent: ${name}`,
78
+ });
79
+ }
80
+ return actions;
36
81
  }
@@ -1,7 +1,7 @@
1
1
  export function formatReport(applied, meta) {
2
2
  const count = (s) => applied.filter(a => a.status === s).length;
3
3
  const lines = [];
4
- lines.push('Yoke retrofit (Claude Code):');
4
+ lines.push('Yoke retrofit:');
5
5
  for (const a of applied) {
6
6
  const note = a.backedUp ? ` (backup: ${a.backedUp})` : '';
7
7
  lines.push(` ${a.status.padEnd(11)} ${a.target}${note}`);
@@ -1,43 +1,83 @@
1
1
  import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runAgent, isAgentAvailable, } from '../loop/runner.js';
2
2
  import { resolveIdleMs } from '../loop/run-command.js';
3
+ import { existsSync, mkdirSync, rmSync, rmdirSync } from 'node:fs';
4
+ import { join } from 'node:path';
5
+ import { loadConfig } from '../retrofit/config.js';
6
+ import { readReviewVerdict, reviewVerdictPath } from './verdict.js';
3
7
  // Resolve to the first available agent, preferring a *second* model so the review
4
8
  // is genuinely cross-model. claude last => a Claude-only box degrades to self-review.
5
9
  const RESOLUTION_ORDER = ['codex', 'gemini', 'claude'];
6
10
  export function runReview(targetDir, opts = {}) {
7
11
  const available = opts.isAvailable ?? isAgentAvailable;
12
+ const implementer = opts.implementer ?? loadConfig(targetDir)?.agents[0] ?? 'claude';
8
13
  let reviewer = opts.reviewer;
9
14
  if (reviewer) {
10
15
  if (!available(reviewer)) {
11
16
  console.error(`Reviewer agent CLI "${reviewer}" was not found on PATH. Install it, or pick another with --reviewer=<claude|codex|gemini>.`);
12
17
  return 2;
13
18
  }
19
+ if (reviewer === implementer && !opts.allowSelfReview) {
20
+ console.error(`Reviewer "${reviewer}" is also the implementer. Pick another agent or pass --allow-self-review explicitly.`);
21
+ return 2;
22
+ }
14
23
  }
15
24
  else {
16
- reviewer = RESOLUTION_ORDER.find(a => available(a));
25
+ reviewer = RESOLUTION_ORDER.find(a => a !== implementer && available(a));
26
+ if (!reviewer && opts.allowSelfReview && available(implementer))
27
+ reviewer = implementer;
17
28
  if (!reviewer) {
18
- console.error('No agent CLI (claude|codex|gemini) found on PATH. Install one to run a review.');
29
+ console.error('No independent reviewer CLI is available. Install a second agent, select one with --reviewer, or pass --allow-self-review explicitly.');
19
30
  return 2;
20
31
  }
21
- if (reviewer === 'claude') {
22
- console.log('Note: only Claude is available — this is a self-review, not cross-model.');
23
- }
24
32
  }
25
33
  const scope = opts.base
26
34
  ? `the diff ${opts.base}..HEAD`
27
35
  : 'the uncommitted working-tree changes (working tree + staged)';
28
- const prompt = buildStandaloneReviewPrompt(scope, opts.focus);
36
+ const verdictPath = reviewVerdictPath(targetDir);
37
+ const yokeDir = join(targetDir, '.yoke');
38
+ const createdYokeDir = !existsSync(yokeDir);
39
+ mkdirSync(yokeDir, { recursive: true });
40
+ rmSync(verdictPath, { force: true });
41
+ const prompt = buildStandaloneReviewPrompt(scope, opts.focus, verdictPath);
29
42
  const idleMs = resolveIdleMs(opts.timeoutMinutes, undefined);
30
43
  // Pass the *agent* invocation to the runner so callers (and tests) see the
31
44
  // reviewer command. The default runner adds the watchdog wrapper before exec;
32
45
  // an injected run() gets the raw invocation.
33
- const inv = agentInvocation(reviewer, prompt, targetDir);
34
- console.log(`Reviewing ${scope} with ${reviewer}...`);
46
+ const inv = agentInvocation(reviewer, prompt, targetDir, 'safe');
47
+ const say = opts.json ? console.error : console.log;
48
+ say(`Reviewing ${scope} with ${reviewer}...`);
35
49
  const run = opts.run ?? ((i) => runAgent(buildWatchdogInvocation(i, idleMs)));
36
- const result = run(inv);
37
- if (result.success) {
38
- console.log(`✓ ${reviewer} approved`);
50
+ const processResult = run(inv);
51
+ let verdict;
52
+ try {
53
+ verdict = readReviewVerdict(verdictPath);
54
+ }
55
+ catch (error) {
56
+ say(`✗ ${reviewer} produced no valid verdict (${error.message})${processResult.success ? '' : `; process: ${processResult.summary}`}`);
57
+ if (createdYokeDir) {
58
+ try {
59
+ rmdirSync(yokeDir);
60
+ }
61
+ catch { }
62
+ }
63
+ return 1;
64
+ }
65
+ if (createdYokeDir) {
66
+ try {
67
+ rmdirSync(yokeDir);
68
+ }
69
+ catch { }
70
+ }
71
+ if (opts.json)
72
+ console.log(JSON.stringify({ reviewer, process: processResult, verdict }));
73
+ if (!processResult.success) {
74
+ say(`✗ ${reviewer} process failed (${processResult.summary}); verdict: ${verdict.summary}`);
75
+ return 1;
76
+ }
77
+ if (verdict.approved) {
78
+ say(`✓ ${reviewer} approved: ${verdict.summary}`);
39
79
  return 0;
40
80
  }
41
- console.log(`✗ ${reviewer} found issues (${result.summary})`);
81
+ say(`✗ ${reviewer} rejected: ${verdict.summary}`);
42
82
  return 1;
43
83
  }
@@ -0,0 +1,45 @@
1
+ import { existsSync, readFileSync, rmSync } from 'node:fs';
2
+ import { join, resolve } from 'node:path';
3
+ import { z } from 'zod';
4
+ export const ReviewFindingSchema = z.object({
5
+ severity: z.enum(['blocking', 'warning', 'info']),
6
+ message: z.string().min(1),
7
+ file: z.string().min(1).optional(),
8
+ line: z.number().int().positive().optional(),
9
+ });
10
+ export const ReviewVerdictSchema = z.object({
11
+ approved: z.boolean(),
12
+ summary: z.string().min(1),
13
+ findings: z.array(ReviewFindingSchema),
14
+ });
15
+ export function reviewVerdictPath(targetDir) {
16
+ return resolve(join(targetDir, '.yoke', 'review-verdict.json'));
17
+ }
18
+ export function readReviewVerdict(path) {
19
+ if (!existsSync(path))
20
+ throw new Error(`Review verdict is missing: ${path}`);
21
+ try {
22
+ let value;
23
+ try {
24
+ value = JSON.parse(readFileSync(path, 'utf8'));
25
+ }
26
+ catch (error) {
27
+ throw new Error(`Review verdict is malformed JSON: ${error.message}`);
28
+ }
29
+ const result = ReviewVerdictSchema.safeParse(value);
30
+ if (!result.success)
31
+ throw new Error(`Review verdict is invalid: ${result.error.message}`);
32
+ return result.data;
33
+ }
34
+ finally {
35
+ rmSync(path, { force: true });
36
+ }
37
+ }
38
+ export function formatReviewContract(path) {
39
+ return [
40
+ `Write your final verdict to this absolute path: ${path}`,
41
+ 'The file must contain exactly one JSON object with this contract:',
42
+ '{"approved":boolean,"summary":"non-empty string","findings":[{"severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1}]}',
43
+ 'Set approved=false when any blocking finding exists. Create the file even when the process also exits non-zero.',
44
+ ].join('\n');
45
+ }