@hecer/yoke 0.8.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codex-plugin/plugin.json +7 -0
- package/CHANGELOG.md +169 -130
- package/README.md +61 -21
- package/TODOS.md +8 -0
- package/agents/docs.toml +6 -0
- package/agents/implementer.toml +6 -0
- package/agents/reviewer.toml +6 -0
- package/agents/security.toml +6 -0
- package/bench/README.md +45 -42
- package/bench/RESULTS.md +46 -36
- package/bench/result-schema.mjs +12 -0
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -0
- package/bench/results/codex-unavailable-1785175418318.json +15 -0
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -0
- package/bench/run-matrix.mjs +26 -0
- package/bench/run.mjs +127 -115
- package/canon/loop/prd.schema.md +5 -0
- package/canon/manifest.yaml +3 -1
- package/canon/skills/authoring-prd/SKILL.md +14 -0
- package/canon/skills/performance/SKILL.md +48 -0
- package/canon/skills/ship/SKILL.md +2 -7
- package/canon/tools/codex-rtk-hook.mjs +36 -0
- package/dist/agents/providers.js +23 -0
- package/dist/agents/telemetry.js +30 -0
- package/dist/agents/types.js +1 -0
- package/dist/audit/changes.js +6 -0
- package/dist/audit/command.js +64 -0
- package/dist/audit/dependencies.js +21 -0
- package/dist/audit/secrets.js +16 -0
- package/dist/audit/types.js +1 -0
- package/dist/cli.js +22 -4
- package/dist/loop/claims.js +57 -0
- package/dist/loop/cleanup.js +10 -4
- package/dist/loop/git.js +8 -2
- package/dist/loop/identity.js +27 -0
- package/dist/loop/loop.js +55 -26
- package/dist/loop/merge-queue.js +20 -0
- package/dist/loop/parallel.js +39 -0
- package/dist/loop/prd.js +48 -2
- package/dist/loop/run-command.js +63 -7
- package/dist/loop/runner.js +47 -31
- package/dist/loop/scheduler.js +8 -0
- package/dist/prd/command.js +6 -0
- package/dist/retrofit/config.js +15 -0
- package/dist/retrofit/planners/codex.js +64 -19
- package/dist/review/command.js +52 -12
- package/dist/review/verdict.js +45 -0
- package/docs/MIGRATING-TO-1.0.md +33 -0
- package/docs/superpowers/plans/2026-07-27-yoke-1.0-release.md +205 -0
- package/docs/superpowers/specs/2026-07-27-yoke-1.0-hardening-and-codex-parity-design.md +164 -0
- package/hooks/hooks.json +19 -0
- package/package.json +82 -67
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/config.yaml +0 -6
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/context/DECISIONS.md +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/.yoke/prd.yaml +0 -38
- package/bench/.runs/claude-2026-07-09T22-34-01/bench-verify.mjs +0 -15
- package/bench/.runs/claude-2026-07-09T22-34-01/package.json +0 -9
- package/bench/.runs/claude-2026-07-09T22-34-01/src/index.mjs +0 -48
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/claude-2026-07-09T22-34-01/tests/STORY-3.test.mjs +0 -25
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/config.yaml +0 -6
- package/bench/.runs/gemini-2026-07-09T22-34-02/.yoke/prd.yaml +0 -32
- package/bench/.runs/gemini-2026-07-09T22-34-02/bench-verify.mjs +0 -15
- package/bench/.runs/gemini-2026-07-09T22-34-02/package.json +0 -9
- package/bench/.runs/gemini-2026-07-09T22-34-02/src/index.mjs +0 -3
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-1.test.mjs +0 -24
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-2.test.mjs +0 -28
- package/bench/.runs/gemini-2026-07-09T22-34-02/tests/STORY-3.test.mjs +0 -25
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
export class MergeQueue {
|
|
2
|
+
tail = Promise.resolve();
|
|
3
|
+
enqueue(job) {
|
|
4
|
+
const run = async () => {
|
|
5
|
+
try {
|
|
6
|
+
await job.rebase();
|
|
7
|
+
if (!await job.verify())
|
|
8
|
+
return { storyId: job.storyId, integrated: false, reason: 'integrated-tree verification failed' };
|
|
9
|
+
await job.integrate();
|
|
10
|
+
return { storyId: job.storyId, integrated: true };
|
|
11
|
+
}
|
|
12
|
+
catch (error) {
|
|
13
|
+
return { storyId: job.storyId, integrated: false, reason: error.message };
|
|
14
|
+
}
|
|
15
|
+
};
|
|
16
|
+
const result = this.tail.then(run, run);
|
|
17
|
+
this.tail = result.then(() => undefined);
|
|
18
|
+
return result;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { readyStories } from './scheduler.js';
|
|
2
|
+
export async function runParallelLoop(stories, opts) {
|
|
3
|
+
const completed = [];
|
|
4
|
+
const failed = [];
|
|
5
|
+
const activeAreas = new Set();
|
|
6
|
+
const active = new Map();
|
|
7
|
+
let iterations = 0;
|
|
8
|
+
let agentIndex = 0;
|
|
9
|
+
const launch = (story) => {
|
|
10
|
+
iterations++;
|
|
11
|
+
if (story.area)
|
|
12
|
+
activeAreas.add(story.area);
|
|
13
|
+
const affinity = story.agent ?? (opts.agents?.length ? opts.agents[agentIndex++ % opts.agents.length] : undefined);
|
|
14
|
+
const promise = opts.worker(story, affinity).then(result => {
|
|
15
|
+
if (result.success) {
|
|
16
|
+
story.passes = true;
|
|
17
|
+
completed.push(story.id);
|
|
18
|
+
}
|
|
19
|
+
else
|
|
20
|
+
failed.push(story.id);
|
|
21
|
+
}).catch(() => { failed.push(story.id); }).finally(() => {
|
|
22
|
+
active.delete(story.id);
|
|
23
|
+
if (story.area)
|
|
24
|
+
activeAreas.delete(story.area);
|
|
25
|
+
});
|
|
26
|
+
active.set(story.id, promise);
|
|
27
|
+
};
|
|
28
|
+
while (iterations < opts.maxIterations && !opts.paused?.()) {
|
|
29
|
+
const slots = Math.max(0, opts.maxConcurrency - active.size);
|
|
30
|
+
const ready = readyStories(stories, { activeAreas }).filter(s => !active.has(s.id) && !failed.includes(s.id)).slice(0, slots);
|
|
31
|
+
for (const story of ready)
|
|
32
|
+
launch(story);
|
|
33
|
+
if (active.size === 0)
|
|
34
|
+
break;
|
|
35
|
+
await Promise.race(active.values());
|
|
36
|
+
}
|
|
37
|
+
await Promise.all(active.values());
|
|
38
|
+
return { completed, failed, iterations, paused: opts.paused?.() ?? false };
|
|
39
|
+
}
|
package/dist/loop/prd.js
CHANGED
|
@@ -7,20 +7,66 @@ export const StorySchema = z.object({
|
|
|
7
7
|
priority: z.number(),
|
|
8
8
|
acceptance: z.array(z.string().min(1)),
|
|
9
9
|
passes: z.boolean(),
|
|
10
|
+
needs: z.array(z.string().min(1)).optional(),
|
|
11
|
+
area: z.string().min(1).optional(),
|
|
12
|
+
agent: z.enum(['claude', 'codex', 'gemini']).optional(),
|
|
10
13
|
});
|
|
11
14
|
const PrdSchema = z.array(StorySchema);
|
|
12
15
|
export function loadPrd(file) {
|
|
13
|
-
|
|
16
|
+
const stories = PrdSchema.parse(parse(readFileSync(file, 'utf8')));
|
|
17
|
+
const issues = validateDependencies(stories);
|
|
18
|
+
if (issues.length)
|
|
19
|
+
throw new Error(`Invalid PRD dependency graph:\n${issues.join('\n')}`);
|
|
20
|
+
return stories;
|
|
14
21
|
}
|
|
15
22
|
export function savePrd(file, stories) {
|
|
16
23
|
writeFileSync(file, stringify(stories));
|
|
17
24
|
}
|
|
18
25
|
export function selectNextStory(stories) {
|
|
19
|
-
const
|
|
26
|
+
const passed = new Set(stories.filter(s => s.passes).map(s => s.id));
|
|
27
|
+
const open = stories.filter(s => !s.passes && (s.needs ?? []).every(id => passed.has(id)));
|
|
20
28
|
if (open.length === 0)
|
|
21
29
|
return null;
|
|
22
30
|
return open.reduce((best, s) => (s.priority < best.priority ? s : best));
|
|
23
31
|
}
|
|
32
|
+
export function validateDependencies(stories) {
|
|
33
|
+
const issues = [];
|
|
34
|
+
const counts = new Map();
|
|
35
|
+
for (const story of stories)
|
|
36
|
+
counts.set(story.id, (counts.get(story.id) ?? 0) + 1);
|
|
37
|
+
for (const [id, count] of counts)
|
|
38
|
+
if (count > 1)
|
|
39
|
+
issues.push(`duplicate story id: ${id}`);
|
|
40
|
+
const ids = new Set(stories.map(s => s.id));
|
|
41
|
+
for (const story of stories) {
|
|
42
|
+
for (const need of story.needs ?? []) {
|
|
43
|
+
if (need === story.id)
|
|
44
|
+
issues.push(`story ${story.id} depends on itself`);
|
|
45
|
+
else if (!ids.has(need))
|
|
46
|
+
issues.push(`story ${story.id} has unknown dependency ${need}`);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
const byId = new Map(stories.map(s => [s.id, s]));
|
|
50
|
+
const visiting = new Set();
|
|
51
|
+
const visited = new Set();
|
|
52
|
+
const walk = (id, path) => {
|
|
53
|
+
if (visiting.has(id)) {
|
|
54
|
+
issues.push(`dependency cycle: ${[...path, id].join(' -> ')}`);
|
|
55
|
+
return;
|
|
56
|
+
}
|
|
57
|
+
if (visited.has(id))
|
|
58
|
+
return;
|
|
59
|
+
visiting.add(id);
|
|
60
|
+
for (const need of byId.get(id)?.needs ?? [])
|
|
61
|
+
if (byId.has(need) && need !== id)
|
|
62
|
+
walk(need, [...path, id]);
|
|
63
|
+
visiting.delete(id);
|
|
64
|
+
visited.add(id);
|
|
65
|
+
};
|
|
66
|
+
for (const id of byId.keys())
|
|
67
|
+
walk(id, []);
|
|
68
|
+
return [...new Set(issues)];
|
|
69
|
+
}
|
|
24
70
|
export function allPass(stories) {
|
|
25
71
|
return stories.length > 0 && stories.every(s => s.passes);
|
|
26
72
|
}
|
package/dist/loop/run-command.js
CHANGED
|
@@ -9,6 +9,8 @@ import { commandVerifier, retryingVerifier } from './verify.js';
|
|
|
9
9
|
import { readStatus, makeReporter, fmtDuration } from './reporter.js';
|
|
10
10
|
import { acquireLock, releaseLock } from './lock.js';
|
|
11
11
|
import { maybeAutoUpgrade } from '../update/upgrade.js';
|
|
12
|
+
import { resolveCommitIdentity } from './identity.js';
|
|
13
|
+
import { runAudit } from '../audit/command.js';
|
|
12
14
|
export const DEFAULT_IDLE_MINUTES = 20;
|
|
13
15
|
const STALE_MINUTES = 20; // a running status older than this likely means the loop died
|
|
14
16
|
export function relativeTime(fromIso, now) {
|
|
@@ -66,6 +68,10 @@ export function resolveIdleMs(flagMinutes, configMinutes) {
|
|
|
66
68
|
return minutes > 0 ? minutes * 60_000 : 0;
|
|
67
69
|
}
|
|
68
70
|
export function runLoopCommand(targetDir, opts) {
|
|
71
|
+
if ((opts.parallel ?? 1) > 1) {
|
|
72
|
+
console.error('Parallel CLI workers are not enabled yet. The dependency-aware dispatcher and merge queue are available as APIs; use --parallel=1 for the synchronous provider runner.');
|
|
73
|
+
return 2;
|
|
74
|
+
}
|
|
69
75
|
const config = loadConfig(targetDir);
|
|
70
76
|
if (!config?.loop.enabled) {
|
|
71
77
|
console.error('Loop is disabled. Enable it with: yoke loop on');
|
|
@@ -85,12 +91,41 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
85
91
|
}
|
|
86
92
|
verify = retryingVerifier(commandVerifier(command), config.verify?.retries ?? 1);
|
|
87
93
|
}
|
|
94
|
+
// Optional performance budget gate: same contract as verify (exit 0 = within
|
|
95
|
+
// budget), same flake tolerance (benchmarks are noisy).
|
|
96
|
+
let perf = opts.perf;
|
|
97
|
+
if (!perf && config.perf?.command) {
|
|
98
|
+
perf = retryingVerifier(commandVerifier(config.perf.command), config.perf.retries ?? 1);
|
|
99
|
+
}
|
|
88
100
|
// Opt-in self-update, loop START only — this run keeps executing the version
|
|
89
101
|
// it started with; a fetched upgrade applies from the next invocation.
|
|
90
102
|
maybeAutoUpgrade(config.update?.auto);
|
|
91
103
|
const available = opts.isAvailable ?? isAgentAvailable;
|
|
92
104
|
const runnerAgent = opts.agent ?? config.agents[0] ?? 'claude';
|
|
105
|
+
const git = opts.git ?? realGitOps;
|
|
106
|
+
let commitIdentity = opts.commitIdentity;
|
|
107
|
+
if (!commitIdentity && !opts.git) {
|
|
108
|
+
try {
|
|
109
|
+
commitIdentity = resolveCommitIdentity(targetDir, config.commit);
|
|
110
|
+
}
|
|
111
|
+
catch (error) {
|
|
112
|
+
console.error(error.message);
|
|
113
|
+
return 2;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
let audit = opts.audit;
|
|
117
|
+
if (!audit && config.audit?.enabled) {
|
|
118
|
+
audit = (dir) => {
|
|
119
|
+
const result = runAudit(dir, { command: config.audit?.command, suppressions: config.audit?.suppressions });
|
|
120
|
+
return { passed: result.code === 0, summary: result.error ?? (result.findings.map(f => `${f.ruleId} ${f.file}${f.line ? `:${f.line}` : ''}`).join(', ') || 'audit passed') };
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
if (commitIdentity) {
|
|
124
|
+
const announce = opts.json ? console.error : console.log;
|
|
125
|
+
announce(`Commits: ${commitIdentity.authorName} <${commitIdentity.authorEmail}> · co-authors: ${commitIdentity.allowCoAuthors ? 'allowed' : 'disabled'}`);
|
|
126
|
+
}
|
|
93
127
|
const idleMs = resolveIdleMs(opts.timeoutMinutes, config.loop.timeoutMinutes);
|
|
128
|
+
const permissions = opts.permissions ?? config.runner?.permissions ?? 'safe';
|
|
94
129
|
let runner = opts.runner;
|
|
95
130
|
if (!runner) {
|
|
96
131
|
if (!available(runnerAgent)) {
|
|
@@ -99,16 +134,34 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
99
134
|
}
|
|
100
135
|
// Token reporting is part of the machine interface: in --json mode a claude
|
|
101
136
|
// runner switches to stream-json so cumulative usage rides on every status.
|
|
102
|
-
runner = makeRunner(runnerAgent, idleMs, {
|
|
137
|
+
runner = makeRunner(runnerAgent, idleMs, {
|
|
138
|
+
tokenReport: opts.json === true,
|
|
139
|
+
onAmbiguity: opts.onAmbiguity ?? config.loop.onAmbiguity,
|
|
140
|
+
perfCommand: config.perf?.command,
|
|
141
|
+
permissions,
|
|
142
|
+
});
|
|
143
|
+
const announce = opts.json ? console.error : console.log;
|
|
144
|
+
announce(`Runner: ${runnerAgent} · permissions: ${permissions} · cwd: ${targetDir}`);
|
|
103
145
|
}
|
|
104
146
|
let review = opts.reviewRunner;
|
|
105
147
|
if (!review && (opts.review || opts.reviewer)) {
|
|
106
|
-
const reviewerAgent = opts.reviewer ?? runnerAgent;
|
|
107
|
-
if (!
|
|
108
|
-
|
|
148
|
+
const reviewerAgent = opts.reviewer ?? ['codex', 'gemini', 'claude'].find(agent => agent !== runnerAgent && available(agent));
|
|
149
|
+
if (!reviewerAgent) {
|
|
150
|
+
if (!opts.allowSelfReview) {
|
|
151
|
+
console.error('No independent reviewer CLI is available. Install or select a second agent, or pass --allow-self-review explicitly.');
|
|
152
|
+
return 2;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
const resolvedReviewer = reviewerAgent ?? runnerAgent;
|
|
156
|
+
if (resolvedReviewer === runnerAgent && !opts.allowSelfReview) {
|
|
157
|
+
console.error(`Reviewer "${resolvedReviewer}" is also the implementer. Pick another agent or pass --allow-self-review explicitly.`);
|
|
158
|
+
return 2;
|
|
159
|
+
}
|
|
160
|
+
if (!available(resolvedReviewer)) {
|
|
161
|
+
console.error(`Reviewer agent CLI "${resolvedReviewer}" was not found on PATH. Install it, or pick another with --reviewer=<claude|codex|gemini>.`);
|
|
109
162
|
return 2;
|
|
110
163
|
}
|
|
111
|
-
review = makeReviewRunner(
|
|
164
|
+
review = makeReviewRunner(resolvedReviewer, idleMs);
|
|
112
165
|
}
|
|
113
166
|
const lock = acquireLock(targetDir);
|
|
114
167
|
if (!lock.acquired) {
|
|
@@ -123,10 +176,13 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
123
176
|
prdPath: path,
|
|
124
177
|
targetDir,
|
|
125
178
|
runner,
|
|
126
|
-
git
|
|
179
|
+
git,
|
|
180
|
+
commitIdentity,
|
|
127
181
|
verify,
|
|
182
|
+
perf,
|
|
183
|
+
audit,
|
|
128
184
|
maxIterations: opts.maxIterations,
|
|
129
|
-
isolate: opts.isolate ?? false,
|
|
185
|
+
isolate: (opts.parallel ?? 1) > 1 ? true : (opts.isolate ?? false),
|
|
130
186
|
review,
|
|
131
187
|
reporter: opts.reporter ?? makeReporter(targetDir, { json: opts.json }),
|
|
132
188
|
});
|
package/dist/loop/runner.js
CHANGED
|
@@ -1,12 +1,15 @@
|
|
|
1
1
|
import { execFileSync, execSync } from 'node:child_process';
|
|
2
|
-
import { existsSync } from 'node:fs';
|
|
2
|
+
import { existsSync, mkdirSync, rmSync } from 'node:fs';
|
|
3
3
|
import { join } from 'node:path';
|
|
4
4
|
import { fileURLToPath } from 'node:url';
|
|
5
5
|
import { loadContext, formatForPrompt, contextDir } from '../context/context.js';
|
|
6
|
+
import { buildProviderInvocation } from '../agents/providers.js';
|
|
7
|
+
import { parseProviderTelemetry } from '../agents/telemetry.js';
|
|
8
|
+
import { formatReviewContract, readReviewVerdict, reviewVerdictPath } from '../review/verdict.js';
|
|
6
9
|
export function contextBlockFor(targetDir) {
|
|
7
10
|
return formatForPrompt(loadContext(contextDir(targetDir)));
|
|
8
11
|
}
|
|
9
|
-
export function buildClaudePrompt(story, context, onAmbiguity = 'resolve') {
|
|
12
|
+
export function buildClaudePrompt(story, context, onAmbiguity = 'resolve', perfCommand) {
|
|
10
13
|
const criteria = story.acceptance.map(a => `- ${a}`).join('\n');
|
|
11
14
|
const lines = [
|
|
12
15
|
'You are an autonomous coding agent running inside the Yoke loop.',
|
|
@@ -16,10 +19,14 @@ export function buildClaudePrompt(story, context, onAmbiguity = 'resolve') {
|
|
|
16
19
|
lines.push('', context);
|
|
17
20
|
lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria (Definition of Done):', criteria, '', "When done, ensure the project's full test suite passes.", 'Do NOT commit — the loop commits on your behalf after verifying.', '', 'Working rules:', '- Add nothing beyond what the story requires: no extra features, abstractions, comments, or defensive code for cases that cannot happen.', '- Do not create summary, plan, or analysis documents — only files the story itself needs.', '- If a check fails, fix the root cause; never bypass it (e.g. --no-verify) or pass by weakening tests.', '- Report the outcome faithfully: if a criterion is unmet or tests fail, say so plainly instead of claiming success.', '- Never ask questions or wait for input — you run unattended and nobody can answer.', onAmbiguity === 'abort'
|
|
18
21
|
? '- If an acceptance criterion is genuinely undecidable, do NOT guess: write the open question(s) to .yoke/ambiguity.md, change nothing else, and stop.'
|
|
19
|
-
: '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.'
|
|
22
|
+
: '- If an acceptance criterion is ambiguous, resolve it yourself in the way most consistent with the other criteria and the existing code, and state your interpretation in your final message.');
|
|
23
|
+
if (perfCommand) {
|
|
24
|
+
lines.push(`- This project enforces a performance budget: \`${perfCommand}\` must exit 0 or the story is blocked. Keep hot paths efficient, and never simplify away an existing optimization without re-running that benchmark.`);
|
|
25
|
+
}
|
|
26
|
+
lines.push('- Keep your final message to a few short sentences: what changed and what you verified.');
|
|
20
27
|
return lines.join('\n');
|
|
21
28
|
}
|
|
22
|
-
export function buildReviewPrompt(story, context) {
|
|
29
|
+
export function buildReviewPrompt(story, context, verdictPath) {
|
|
23
30
|
const criteria = story.acceptance.map(a => `- ${a}`).join('\n');
|
|
24
31
|
const lines = [
|
|
25
32
|
'You are an independent reviewer inside the Yoke loop. You did NOT implement this change.',
|
|
@@ -27,10 +34,12 @@ export function buildReviewPrompt(story, context) {
|
|
|
27
34
|
];
|
|
28
35
|
if (context)
|
|
29
36
|
lines.push('', context);
|
|
30
|
-
lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve
|
|
37
|
+
lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
|
|
38
|
+
if (verdictPath)
|
|
39
|
+
lines.push('', formatReviewContract(verdictPath));
|
|
31
40
|
return lines.join('\n');
|
|
32
41
|
}
|
|
33
|
-
export function buildStandaloneReviewPrompt(scope, focus) {
|
|
42
|
+
export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
|
|
34
43
|
const lines = [
|
|
35
44
|
'You are an independent reviewer. You did NOT write this change.',
|
|
36
45
|
`Review ${scope}. Run git yourself to see the diff (e.g. \`git diff\`, or \`git diff <base>..HEAD\`).`,
|
|
@@ -39,6 +48,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
|
|
|
39
48
|
if (focus)
|
|
40
49
|
lines.push(`Pay particular attention to: ${focus}.`);
|
|
41
50
|
lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
|
|
51
|
+
if (verdictPath)
|
|
52
|
+
lines.push('', formatReviewContract(verdictPath));
|
|
42
53
|
return lines.join('\n');
|
|
43
54
|
}
|
|
44
55
|
// Headless agents must run non-interactively: with plain `-p` the CLI denies
|
|
@@ -47,16 +58,8 @@ export function buildStandaloneReviewPrompt(scope, focus) {
|
|
|
47
58
|
// and falsely marks the story done. Granting autonomous permissions makes the
|
|
48
59
|
// implementer actually able to write files and run the verify command.
|
|
49
60
|
// (The loop is opt-in and scoped to the target project dir.)
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
codex: { command: 'codex', baseArgs: ['exec', '--dangerously-bypass-approvals-and-sandbox'] },
|
|
53
|
-
// gemini: no `-p` — current Gemini CLI (0.33+) requires a value after -p, and
|
|
54
|
-
// piped (non-TTY) stdin already selects headless mode on its own.
|
|
55
|
-
gemini: { command: 'gemini', baseArgs: ['--yolo'] },
|
|
56
|
-
};
|
|
57
|
-
export function agentInvocation(agent, prompt, cwd) {
|
|
58
|
-
const spec = AGENT_SPECS[agent];
|
|
59
|
-
return { command: spec.command, args: spec.baseArgs, input: prompt, cwd };
|
|
61
|
+
export function agentInvocation(agent, prompt, cwd, permissions = 'safe') {
|
|
62
|
+
return buildProviderInvocation(agent, prompt, cwd, permissions);
|
|
60
63
|
}
|
|
61
64
|
export function claudeInvocation(prompt, cwd) {
|
|
62
65
|
return agentInvocation('claude', prompt, cwd);
|
|
@@ -65,7 +68,7 @@ export function claudeInvocation(prompt, cwd) {
|
|
|
65
68
|
// (--verbose is required by the CLI for stream-json in -p mode). Prompt still via stdin.
|
|
66
69
|
// Derived from the base spec so the headless permission-bypass flag rides along.
|
|
67
70
|
export function claudeStreamJsonInvocation(prompt, cwd) {
|
|
68
|
-
return
|
|
71
|
+
return buildProviderInvocation('claude', prompt, cwd, 'safe');
|
|
69
72
|
}
|
|
70
73
|
// Pick the runner invocation. Claude ALWAYS runs in stream-json mode: plain `-p`
|
|
71
74
|
// prints nothing until the run finishes, so the idle watchdog saw a healthy
|
|
@@ -73,10 +76,8 @@ export function claudeStreamJsonInvocation(prompt, cwd) {
|
|
|
73
76
|
// the user saw dead air the whole time. stream-json emits per-message output,
|
|
74
77
|
// which doubles as liveness. Token usage rides along for free. Other agents
|
|
75
78
|
// keep their plain invocation (no machine-readable stream to gain).
|
|
76
|
-
export function runnerInvocation(agent, prompt, cwd, _tokenReport = false) {
|
|
77
|
-
|
|
78
|
-
return claudeStreamJsonInvocation(prompt, cwd);
|
|
79
|
-
return agentInvocation(agent, prompt, cwd);
|
|
79
|
+
export function runnerInvocation(agent, prompt, cwd, _tokenReport = false, permissions = 'safe') {
|
|
80
|
+
return buildProviderInvocation(agent, prompt, cwd, permissions);
|
|
80
81
|
}
|
|
81
82
|
// Parse claude stream-json output into cumulative token usage. Defensive by design:
|
|
82
83
|
// non-JSON lines and unknown message shapes are ignored. The final "result" message
|
|
@@ -226,20 +227,21 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
|
|
|
226
227
|
// Claude always streams (see runnerInvocation) — capture the stream so tokens are
|
|
227
228
|
// always reported; other agents keep inherit stdio. opts.tokenReport is now
|
|
228
229
|
// redundant for claude and meaningless elsewhere; kept for caller compatibility.
|
|
229
|
-
const captureTokens =
|
|
230
|
+
const captureTokens = true;
|
|
230
231
|
return (ctx) => {
|
|
231
|
-
const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity), ctx.targetDir, captureTokens);
|
|
232
|
+
const base = runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, captureTokens, opts.permissions ?? 'safe');
|
|
232
233
|
const inv = buildWatchdogInvocation(base, idleTimeoutMs);
|
|
233
234
|
if (captureTokens) {
|
|
234
235
|
const capture = opts.execCapture ?? runCliCapture;
|
|
235
236
|
try {
|
|
236
237
|
const out = capture(inv);
|
|
237
|
-
|
|
238
|
+
const telemetry = parseProviderTelemetry(agent, out.split(/\r?\n/));
|
|
239
|
+
return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: telemetry.tokens };
|
|
238
240
|
}
|
|
239
241
|
catch (e) {
|
|
240
242
|
// Salvage usage from whatever the agent streamed before dying — those tokens were spent.
|
|
241
243
|
const partial = e.stdout;
|
|
242
|
-
const tokens = partial == null ? undefined :
|
|
244
|
+
const tokens = partial == null ? undefined : parseProviderTelemetry(agent, String(partial).split(/\r?\n/)).tokens;
|
|
243
245
|
return { success: false, summary: `${agent} failed on ${ctx.story.id}: ${e.message}`, tokens };
|
|
244
246
|
}
|
|
245
247
|
}
|
|
@@ -256,21 +258,35 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
|
|
|
256
258
|
};
|
|
257
259
|
}
|
|
258
260
|
export const claudeRunner = makeRunner('claude');
|
|
259
|
-
export function makeReviewRunner(agent, idleTimeoutMs = 0) {
|
|
261
|
+
export function makeReviewRunner(agent, idleTimeoutMs = 0, exec = runCli) {
|
|
260
262
|
return (ctx) => {
|
|
261
|
-
const
|
|
263
|
+
const verdictPath = reviewVerdictPath(ctx.targetDir);
|
|
264
|
+
mkdirSync(join(ctx.targetDir, '.yoke'), { recursive: true });
|
|
265
|
+
rmSync(verdictPath, { force: true });
|
|
266
|
+
const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir), verdictPath), ctx.targetDir, 'safe');
|
|
262
267
|
const inv = buildWatchdogInvocation(base, idleTimeoutMs);
|
|
268
|
+
let processFailure;
|
|
269
|
+
try {
|
|
270
|
+
exec(inv);
|
|
271
|
+
}
|
|
272
|
+
catch (e) {
|
|
273
|
+
processFailure = e.message;
|
|
274
|
+
}
|
|
263
275
|
try {
|
|
264
|
-
|
|
265
|
-
|
|
276
|
+
const verdict = readReviewVerdict(verdictPath);
|
|
277
|
+
if (processFailure)
|
|
278
|
+
return { success: false, summary: `review process failed: ${processFailure}; verdict: ${verdict.summary}` };
|
|
279
|
+
return verdict.approved
|
|
280
|
+
? { success: true, summary: `${agent} approved ${ctx.story.id}: ${verdict.summary}` }
|
|
281
|
+
: { success: false, summary: `${agent} rejected ${ctx.story.id}: ${verdict.summary}` };
|
|
266
282
|
}
|
|
267
283
|
catch (e) {
|
|
268
|
-
return { success: false, summary: `${
|
|
284
|
+
return { success: false, summary: `${processFailure ? `review process failed: ${processFailure}; ` : ''}${e.message}` };
|
|
269
285
|
}
|
|
270
286
|
};
|
|
271
287
|
}
|
|
272
288
|
// Probe whether the agent's CLI is on PATH (so the loop can refuse upfront with a
|
|
273
289
|
// clear message instead of failing mid-run with spawn ENOENT). Never throws.
|
|
274
290
|
export function isAgentAvailable(agent) {
|
|
275
|
-
return probeVersion(
|
|
291
|
+
return probeVersion(agent);
|
|
276
292
|
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
export function readyStories(stories, opts = {}) {
|
|
2
|
+
const passed = new Set(stories.filter(story => story.passes).map(story => story.id));
|
|
3
|
+
return stories
|
|
4
|
+
.filter(story => !story.passes)
|
|
5
|
+
.filter(story => (story.needs ?? []).every(id => passed.has(id)))
|
|
6
|
+
.filter(story => !story.area || !opts.activeAreas?.has(story.area))
|
|
7
|
+
.sort((a, b) => a.priority - b.priority || Number(b.agent === opts.agent) - Number(a.agent === opts.agent) || a.id.localeCompare(b.id));
|
|
8
|
+
}
|
package/dist/prd/command.js
CHANGED
|
@@ -9,6 +9,9 @@ export const PRD_TEMPLATE = `# Yoke PRD — the loop picks the lowest-priority o
|
|
|
9
9
|
# - id: STORY-1
|
|
10
10
|
# title: scaffold the project with a runnable test suite
|
|
11
11
|
# priority: 1
|
|
12
|
+
# needs: [] # optional story IDs that must pass first
|
|
13
|
+
# area: foundation # optional collision domain for parallel runs
|
|
14
|
+
# agent: codex # optional claude|codex|gemini affinity
|
|
12
15
|
# acceptance:
|
|
13
16
|
# - "the verify command exits 0"
|
|
14
17
|
# - "a placeholder test exists and passes"
|
|
@@ -26,6 +29,9 @@ export function buildPrdDraftPrompt(idea) {
|
|
|
26
29
|
'- id: STORY-1, STORY-2, ... (unique)',
|
|
27
30
|
'- title: one imperative sentence',
|
|
28
31
|
'- priority: dense integers from 1 (lower = built first)',
|
|
32
|
+
'- needs: optional list of story IDs that must pass first; the graph must be acyclic',
|
|
33
|
+
'- area: optional collision domain for safe parallel scheduling',
|
|
34
|
+
'- agent: optional claude|codex|gemini affinity',
|
|
29
35
|
'- acceptance: 2-5 testable, behavioral criteria (observable outcomes, never implementation steps)',
|
|
30
36
|
'- passes: false',
|
|
31
37
|
'',
|
package/dist/retrofit/config.js
CHANGED
|
@@ -16,7 +16,22 @@ export const YokeConfigSchema = z.object({
|
|
|
16
16
|
// or 'abort' (agent stops the story via .yoke/ambiguity.md for a human decision).
|
|
17
17
|
onAmbiguity: z.enum(['resolve', 'abort']).optional(),
|
|
18
18
|
}),
|
|
19
|
+
runner: z.object({ permissions: z.enum(['safe', 'unsafe', 'read-only']) }).optional(),
|
|
20
|
+
commit: z.object({
|
|
21
|
+
authorName: z.string().min(1).optional(),
|
|
22
|
+
authorEmail: z.string().email().optional(),
|
|
23
|
+
allowCoAuthors: z.boolean().optional(),
|
|
24
|
+
}).optional(),
|
|
25
|
+
audit: z.object({
|
|
26
|
+
enabled: z.boolean(),
|
|
27
|
+
command: z.string().min(1).optional(),
|
|
28
|
+
suppressionsVersion: z.literal(1).optional(),
|
|
29
|
+
suppressions: z.array(z.object({ ruleId: z.string().min(1), file: z.string().min(1).optional(), reason: z.string(), expires: z.string().optional() })).optional(),
|
|
30
|
+
}).optional(),
|
|
19
31
|
verify: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
|
|
32
|
+
// Optional performance budget gate: a benchmark command that must exit 0 for a
|
|
33
|
+
// story to land (runs after verify). Benchmarks are noisy → retried like verify.
|
|
34
|
+
perf: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
|
|
20
35
|
codeGraph: CodeGraphSchema.optional(),
|
|
21
36
|
smoke: SmokeSchema.optional(),
|
|
22
37
|
// Opt-in: upgrade yoke at loop START when a newer version is cached (never mid-run).
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { readFileSync } from 'node:fs';
|
|
2
2
|
import { join } from 'node:path';
|
|
3
|
+
import { loadManifest } from '../../canon/manifest.js';
|
|
3
4
|
import { mcpServers, rtkInstruction } from '../tools.js';
|
|
4
5
|
function tomlMcp(codeGraph) {
|
|
5
6
|
const servers = mcpServers(codeGraph);
|
|
@@ -13,24 +14,68 @@ function tomlMcp(codeGraph) {
|
|
|
13
14
|
.join('\n');
|
|
14
15
|
}
|
|
15
16
|
export function planCodex(canonDir, _targetDir, codeGraph = 'graphify') {
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
{
|
|
30
|
-
kind: 'write',
|
|
31
|
-
target: 'RTK.md',
|
|
32
|
-
content: rtkInstruction() + '\n',
|
|
33
|
-
reason: 'rtk instruction (Codex has no rewrite hook)',
|
|
34
|
-
},
|
|
17
|
+
const manifest = loadManifest(join(canonDir, 'manifest.yaml'));
|
|
18
|
+
const baseline = readFileSync(join(canonDir, 'AGENTS.md'), 'utf8');
|
|
19
|
+
const actions = manifest.skills.map(skill => ({
|
|
20
|
+
kind: 'write',
|
|
21
|
+
target: `.agents/skills/${skill.id}/SKILL.md`,
|
|
22
|
+
content: readFileSync(join(canonDir, skill.path, 'SKILL.md'), 'utf8'),
|
|
23
|
+
reason: `skill: ${skill.id}`,
|
|
24
|
+
}));
|
|
25
|
+
const roles = [
|
|
26
|
+
['implementer', 'Implementation specialist for one scoped story.', 'workspace-write', 'Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.'],
|
|
27
|
+
['reviewer', 'Read-only reviewer for correctness and acceptance criteria.', 'read-only', 'Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.'],
|
|
28
|
+
['security', 'Read-only security reviewer for changed code.', 'read-only', 'Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.'],
|
|
29
|
+
['docs', 'Documentation specialist for release and API consistency.', 'workspace-write', 'Update only documentation required by the assigned change. Verify commands and version references against the repository.'],
|
|
35
30
|
];
|
|
31
|
+
actions.push({
|
|
32
|
+
kind: 'write',
|
|
33
|
+
target: 'AGENTS.md',
|
|
34
|
+
content: `${baseline.trimEnd()}\n\n@RTK.md\n`,
|
|
35
|
+
reason: 'baseline instructions (Codex reads AGENTS.md natively)',
|
|
36
|
+
}, {
|
|
37
|
+
kind: 'write',
|
|
38
|
+
target: '.codex/config.toml',
|
|
39
|
+
content: `# Yoke project configuration. Codex loads this in trusted repositories.\n\n[features]\nhooks = true\n\n${tomlMcp(codeGraph)}`,
|
|
40
|
+
reason: 'MCP servers (code-graph + playwright)',
|
|
41
|
+
}, {
|
|
42
|
+
kind: 'write',
|
|
43
|
+
target: '.codex/hooks.json',
|
|
44
|
+
merge: true,
|
|
45
|
+
content: JSON.stringify({
|
|
46
|
+
description: 'Yoke command compression for Codex',
|
|
47
|
+
hooks: {
|
|
48
|
+
PreToolUse: [{
|
|
49
|
+
matcher: '^Bash$',
|
|
50
|
+
hooks: [{
|
|
51
|
+
type: 'command',
|
|
52
|
+
command: 'node "$(git rev-parse --show-toplevel)/.codex/hooks/rtk.mjs"',
|
|
53
|
+
commandWindows: 'powershell -NoProfile -ExecutionPolicy Bypass -Command "$root = git rev-parse --show-toplevel; node (Join-Path $root \'.codex/hooks/rtk.mjs\')"',
|
|
54
|
+
timeout: 5,
|
|
55
|
+
statusMessage: 'Compressing command output with RTK',
|
|
56
|
+
}],
|
|
57
|
+
}],
|
|
58
|
+
},
|
|
59
|
+
}, null, 2) + '\n',
|
|
60
|
+
reason: 'rtk PreToolUse hook adapter',
|
|
61
|
+
}, {
|
|
62
|
+
kind: 'write',
|
|
63
|
+
target: '.codex/hooks/rtk.mjs',
|
|
64
|
+
content: readFileSync(join(canonDir, 'tools', 'codex-rtk-hook.mjs'), 'utf8'),
|
|
65
|
+
reason: 'rtk Codex hook adapter',
|
|
66
|
+
}, {
|
|
67
|
+
kind: 'write',
|
|
68
|
+
target: 'RTK.md',
|
|
69
|
+
content: rtkInstruction() + '\n',
|
|
70
|
+
reason: 'rtk instruction (Codex has no rewrite hook)',
|
|
71
|
+
});
|
|
72
|
+
for (const [name, description, sandbox, instructions] of roles) {
|
|
73
|
+
actions.push({
|
|
74
|
+
kind: 'write',
|
|
75
|
+
target: `.codex/agents/${name}.toml`,
|
|
76
|
+
content: `name = "${name}"\ndescription = "${description}"\nsandbox_mode = "${sandbox}"\ndeveloper_instructions = """\n${instructions}\n"""\n`,
|
|
77
|
+
reason: `Codex role agent: ${name}`,
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
return actions;
|
|
36
81
|
}
|