@hecer/yoke 1.2.1 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +66 -24
- package/README.md +186 -104
- package/TODOS.md +0 -3
- package/canon/loop/loop-spec.md +53 -24
- package/canon/loop/prd.schema.md +30 -6
- package/canon/manifest.yaml +1 -1
- package/canon/skills/authoring-prd/SKILL.md +30 -31
- package/dist/agents/contracts.js +50 -0
- package/dist/agents/process-incarnation.js +15 -0
- package/dist/agents/process-record.js +65 -0
- package/dist/agents/process-streams.js +40 -0
- package/dist/agents/process.js +177 -0
- package/dist/agents/providers.js +10 -7
- package/dist/agents/telemetry.js +62 -0
- package/dist/change/inbox.js +279 -0
- package/dist/cli.js +90 -4
- package/dist/loop/candidate-boundaries.js +43 -0
- package/dist/loop/candidate-cleanup.js +98 -0
- package/dist/loop/candidate-contracts.js +1 -0
- package/dist/loop/candidate-selection.js +84 -0
- package/dist/loop/candidates.js +228 -0
- package/dist/loop/claim-lease.js +131 -0
- package/dist/loop/claims.js +177 -40
- package/dist/loop/cleanup.js +117 -15
- package/dist/loop/decision.js +31 -0
- package/dist/loop/dispatcher.js +334 -0
- package/dist/loop/evidence.js +31 -0
- package/dist/loop/gates.js +10 -1
- package/dist/loop/loop.js +236 -33
- package/dist/loop/merge-queue.js +12 -6
- package/dist/loop/parallel-adapters.js +185 -0
- package/dist/loop/parallel-command.js +287 -0
- package/dist/loop/parallel.js +2 -4
- package/dist/loop/prd.js +63 -2
- package/dist/loop/reporter.js +86 -5
- package/dist/loop/run-command.js +227 -53
- package/dist/loop/runner.js +77 -34
- package/dist/loop/verify.js +11 -0
- package/dist/loop/watchdog.js +67 -8
- package/dist/loop/worker-cancellation.js +17 -0
- package/dist/loop/worker-cleanup.js +23 -0
- package/dist/loop/worker-contracts.js +1 -0
- package/dist/loop/worker.js +254 -0
- package/dist/prd/command.js +18 -5
- package/dist/quality/artifacts.js +59 -0
- package/dist/quality/candidate-comparison.js +130 -0
- package/dist/quality/command.js +316 -0
- package/dist/quality/loop.js +86 -0
- package/dist/quality/process-command.js +57 -0
- package/dist/quality/reference.js +187 -0
- package/dist/quality/repair.js +11 -0
- package/dist/quality/runner.js +66 -0
- package/dist/quality/types.js +60 -0
- package/dist/quality/verdict.js +142 -0
- package/dist/retrofit/config.js +13 -4
- package/dist/retrofit/gitignore.js +4 -0
- package/dist/review/command.js +27 -38
- package/dist/review/verdict.js +38 -7
- package/dist/routing/router.js +4 -1
- package/docs/MIGRATING-TO-1.4.md +70 -0
- package/docs/PUBLISHING.md +16 -2
- package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -0
- package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -0
- package/gemini-extension.json +1 -1
- package/package.json +6 -6
package/dist/loop/run-command.js
CHANGED
|
@@ -3,17 +3,21 @@ import { existsSync } from 'node:fs';
|
|
|
3
3
|
import { loadConfig, saveConfig, defaultConfig, resolveVerifyCommand } from '../retrofit/config.js';
|
|
4
4
|
import { loadPrd, progress } from './prd.js';
|
|
5
5
|
import { runLoop } from './loop.js';
|
|
6
|
-
import { realGitOps } from './git.js';
|
|
6
|
+
import { commitPaths, realGitOps } from './git.js';
|
|
7
7
|
import { makeRunner, makeReviewRunner, isAgentAvailable } from './runner.js';
|
|
8
|
-
import { commandVerifier, retryingVerifier } from './verify.js';
|
|
8
|
+
import { commandVerifier, commandsVerifier, retryingVerifier } from './verify.js';
|
|
9
9
|
import { readStatus, makeReporter, fmtDuration } from './reporter.js';
|
|
10
10
|
import { acquireLock, releaseLock } from './lock.js';
|
|
11
11
|
import { maybeAutoUpgrade } from '../update/upgrade.js';
|
|
12
12
|
import { resolveCommitIdentity } from './identity.js';
|
|
13
13
|
import { runAudit } from '../audit/command.js';
|
|
14
14
|
import { detectHostAgent, resolveRunnerAgent } from '../agents/host.js';
|
|
15
|
-
import { clearDecisionResume, decisionProcessingExists, decisionRequestId, formatPendingDecision, readPendingDecision, writeDecisionResume, } from './decision.js';
|
|
15
|
+
import { buildTrustedDecisionResumeState, clearDecisionResume, decisionProcessingExists, decisionRequestId, formatPendingDecision, readPendingDecision, writeDecisionResume, } from './decision.js';
|
|
16
16
|
import { makeAdaptiveRunner } from '../routing/router.js';
|
|
17
|
+
import { runChangeApply } from '../change/inbox.js';
|
|
18
|
+
import { createQualityCommandHooks } from '../quality/command.js';
|
|
19
|
+
import { resolveQualityPolicy } from '../quality/types.js';
|
|
20
|
+
import { runParallelLoopCommand } from './parallel-command.js';
|
|
17
21
|
export const DEFAULT_IDLE_MINUTES = 20;
|
|
18
22
|
const STALE_MINUTES = 20; // a running status older than this likely means the loop died
|
|
19
23
|
export function relativeTime(fromIso, now) {
|
|
@@ -60,6 +64,24 @@ export function loopStatus(targetDir, now = () => new Date()) {
|
|
|
60
64
|
}
|
|
61
65
|
if (st.reason)
|
|
62
66
|
lines.push(` reason: ${st.reason}`);
|
|
67
|
+
if (st.quality)
|
|
68
|
+
lines.push(` quality: round ${st.quality.currentRound} · ${st.quality.usedRepairs}${st.quality.unbounded ? ' unbounded repairs' : `/${st.quality.maxRepairs ?? 0} repairs`} · ${st.quality.policy}`);
|
|
69
|
+
if (st.parallel)
|
|
70
|
+
lines.push(` parallel ${st.parallel.dispatcherId}: ${st.parallel.activeWorkers}/${st.parallel.maxConcurrency} workers · ${st.parallel.queuedCandidates} queued · ${st.parallel.integrated} integrated · ${st.parallel.reopened} reopened`);
|
|
71
|
+
const integrator = st.parallel?.integrator;
|
|
72
|
+
if (integrator) {
|
|
73
|
+
const quality = integrator.quality
|
|
74
|
+
? ` · quality round ${integrator.quality.currentRound} · ${integrator.quality.usedRepairs}${integrator.quality.unbounded ? ' unbounded repairs' : `/${integrator.quality.maxRepairs ?? 0} repairs`}`
|
|
75
|
+
: '';
|
|
76
|
+
lines.push(` integrator ${integrator.story} "${integrator.storyTitle}" (${integrator.provider}${integrator.model ? `/${integrator.model}` : ''}) · ${integrator.phase ?? 'working'}${quality}`);
|
|
77
|
+
}
|
|
78
|
+
for (const worker of st.parallel?.workers ?? []) {
|
|
79
|
+
const quality = worker.quality
|
|
80
|
+
? ` · quality round ${worker.quality.currentRound} · ${worker.quality.usedRepairs}${worker.quality.unbounded ? ' unbounded repairs' : `/${worker.quality.maxRepairs ?? 0} repairs`}`
|
|
81
|
+
: '';
|
|
82
|
+
const candidate = worker.candidateId ? ` candidate ${worker.candidateId} · ${worker.worktree ?? 'worktree unknown'} · ${worker.lifecycle ?? 'working'}` : '';
|
|
83
|
+
lines.push(` worker ${worker.story} "${worker.storyTitle}" (${worker.provider}${worker.model ? `/${worker.model}` : ''})${candidate} · ${worker.phase ?? 'working'}${quality}`);
|
|
84
|
+
}
|
|
63
85
|
const ageMs = now().getTime() - Date.parse(st.updatedAt);
|
|
64
86
|
if (st.state === 'running' && ageMs > STALE_MINUTES * 60_000) {
|
|
65
87
|
lines.push(` ⚠ possibly stuck — no update in ${relativeTime(st.updatedAt, now())}`);
|
|
@@ -71,11 +93,27 @@ export function resolveIdleMs(flagMinutes, configMinutes) {
|
|
|
71
93
|
return minutes > 0 ? minutes * 60_000 : 0;
|
|
72
94
|
}
|
|
73
95
|
export function runLoopCommand(targetDir, opts) {
|
|
74
|
-
|
|
75
|
-
|
|
96
|
+
const parallel = opts.parallel ?? 1;
|
|
97
|
+
const candidates = opts.candidates ?? 1;
|
|
98
|
+
if (!Number.isInteger(parallel) || parallel < 1) {
|
|
99
|
+
console.error('--parallel must be a positive integer');
|
|
100
|
+
return 2;
|
|
101
|
+
}
|
|
102
|
+
if (!Number.isInteger(candidates) || candidates < 1 || candidates > 5) {
|
|
103
|
+
console.error('--candidates must be an integer from 1 to 5');
|
|
76
104
|
return 2;
|
|
77
105
|
}
|
|
78
106
|
const config = loadConfig(targetDir);
|
|
107
|
+
const maxParallelCandidates = config?.quality?.maxParallelCandidates ?? 1;
|
|
108
|
+
if (candidates > maxParallelCandidates) {
|
|
109
|
+
console.error(`--candidates=${candidates} exceeds quality.maxParallelCandidates=${maxParallelCandidates}`);
|
|
110
|
+
return 2;
|
|
111
|
+
}
|
|
112
|
+
const qualityDisabled = opts.quality === false || (!config?.quality?.enabled && opts.quality !== true && opts.qualityUnbounded !== true);
|
|
113
|
+
if (candidates > 1 && qualityDisabled) {
|
|
114
|
+
console.error(`--candidates=${candidates} requires quality; quality cannot be disabled for candidate dispatch.`);
|
|
115
|
+
return 2;
|
|
116
|
+
}
|
|
79
117
|
if (!config?.loop.enabled) {
|
|
80
118
|
console.error('Loop is disabled. Enable it with: yoke loop on');
|
|
81
119
|
return 2;
|
|
@@ -99,6 +137,13 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
99
137
|
console.error(`No PRD found at ${path}. Create one (see canon loop/prd.schema.md).`);
|
|
100
138
|
return 2;
|
|
101
139
|
}
|
|
140
|
+
if (candidates > 1) {
|
|
141
|
+
const missingQuality = loadPrd(path).find(story => !story.passes && !story.quality);
|
|
142
|
+
if (missingQuality) {
|
|
143
|
+
console.error(`Story ${missingQuality.id} needs a quality declaration before --candidates=${candidates} can dispatch.`);
|
|
144
|
+
return 2;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
102
147
|
let verify = opts.verify;
|
|
103
148
|
if (!verify) {
|
|
104
149
|
const command = resolveVerifyCommand(targetDir, config);
|
|
@@ -114,6 +159,9 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
114
159
|
if (!perf && config.perf?.command) {
|
|
115
160
|
perf = retryingVerifier(commandVerifier(config.perf.command), config.perf.retries ?? 1);
|
|
116
161
|
}
|
|
162
|
+
const completion = config.completion?.command
|
|
163
|
+
? retryingVerifier(commandVerifier(config.completion.command), config.completion.retries ?? 1)
|
|
164
|
+
: undefined;
|
|
117
165
|
// Opt-in self-update, loop START only — this run keeps executing the version
|
|
118
166
|
// it started with; a fetched upgrade applies from the next invocation.
|
|
119
167
|
maybeAutoUpgrade(config.update?.auto);
|
|
@@ -142,31 +190,119 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
142
190
|
announce(`Commits: ${commitIdentity.authorName} <${commitIdentity.authorEmail}> · co-authors: ${commitIdentity.allowCoAuthors ? 'allowed' : 'disabled'}`);
|
|
143
191
|
}
|
|
144
192
|
const idleMs = resolveIdleMs(opts.timeoutMinutes, config.loop.timeoutMinutes);
|
|
193
|
+
const qualityOverrides = {
|
|
194
|
+
...(opts.qualityUnbounded ? { quality: true, qualityUnbounded: true } : opts.quality !== undefined ? { quality: opts.quality } : {}),
|
|
195
|
+
...(opts.qualityRounds !== undefined ? { qualityRounds: opts.qualityRounds } : {}),
|
|
196
|
+
...(opts.qualityMinutes !== undefined ? { qualityMinutes: opts.qualityMinutes } : {}),
|
|
197
|
+
...(opts.qualityPolicy ? { qualityPolicy: opts.qualityPolicy } : {}),
|
|
198
|
+
...(opts.candidates !== undefined ? { candidates: opts.candidates } : {}),
|
|
199
|
+
};
|
|
200
|
+
if (qualityOverrides.qualityUnbounded) {
|
|
201
|
+
console.error('WARNING: Quality repair limits are unbounded for this invocation. Mechanical gates, watchdog, isolation, and commit safety remain active.');
|
|
202
|
+
}
|
|
203
|
+
const configuredCriticAgent = config.quality?.critic?.agent ?? config.quality?.criticAgent ?? config.agents.find(agent => agent !== runnerAgent) ?? runnerAgent;
|
|
204
|
+
const configuredCriticModel = config.quality?.critic?.model ?? config.quality?.criticModel ?? (configuredCriticAgent === runnerAgent ? config.runner?.model : undefined);
|
|
205
|
+
if (candidates > 1 && !configuredCriticModel) {
|
|
206
|
+
console.error('Quality candidate selection requires quality.critic.model (or legacy quality.criticModel) before any runner or worktree is started.');
|
|
207
|
+
return 2;
|
|
208
|
+
}
|
|
209
|
+
const quality = createQualityCommandHooks({
|
|
210
|
+
targetDir,
|
|
211
|
+
config,
|
|
212
|
+
runnerAgent,
|
|
213
|
+
idleMs,
|
|
214
|
+
policy: qualityOverrides,
|
|
215
|
+
...(opts.qualityRuntime ? { runtime: opts.qualityRuntime } : {}),
|
|
216
|
+
});
|
|
217
|
+
if (quality) {
|
|
218
|
+
const resolved = resolveQualityPolicy({ defaults: config.quality, overrides: qualityOverrides });
|
|
219
|
+
const criticAgent = configuredCriticAgent;
|
|
220
|
+
const repairAgent = config.quality?.repair?.agent ?? config.quality?.repairAgent ?? runnerAgent;
|
|
221
|
+
const limit = resolved.limits.unbounded ? 'unbounded' : `${resolved.limits.maxRounds ?? 3} rounds/${resolved.limits.maxMinutes ?? 60} minutes`;
|
|
222
|
+
const announce = opts.json ? console.error : console.log;
|
|
223
|
+
announce(`Quality: ${resolved.policy} · critic: ${criticAgent}${configuredCriticModel ? `/${configuredCriticModel}` : '/provider-default'} · repair: ${repairAgent}${config.quality?.repair?.model ?? config.quality?.repairModel ? `/${config.quality?.repair?.model ?? config.quality?.repairModel}` : '/provider-default'} · permissions: read-only critic/safe repair · budget: ${limit}`);
|
|
224
|
+
}
|
|
145
225
|
const permissions = opts.permissions ?? config.runner?.permissions ?? 'safe';
|
|
146
226
|
const routingEnabled = opts.routing ?? config.routing?.enabled ?? false;
|
|
227
|
+
const runnerSelection = {
|
|
228
|
+
model: config.runner?.model,
|
|
229
|
+
reasoningEffort: config.runner?.reasoningEffort,
|
|
230
|
+
bare: config.runner?.bare,
|
|
231
|
+
...((routingEnabled || opts.routing === false) ? { nativeMultiAgent: false } : {}),
|
|
232
|
+
};
|
|
233
|
+
if ((parallel > 1 || candidates > 1) && routingEnabled) {
|
|
234
|
+
console.error('Adaptive routing is not available with parallel workers or quality candidates. Run with --parallel=1 --candidates=1 or disable routing.');
|
|
235
|
+
return 2;
|
|
236
|
+
}
|
|
237
|
+
const parallelProviders = [{
|
|
238
|
+
provider: runnerAgent,
|
|
239
|
+
...(runnerSelection.model ? { model: runnerSelection.model } : {}),
|
|
240
|
+
...(runnerSelection.reasoningEffort ? { reasoningEffort: runnerSelection.reasoningEffort } : {}),
|
|
241
|
+
}];
|
|
242
|
+
const parallelAffinityProviders = (config.routing?.workers ?? []).map(worker => ({
|
|
243
|
+
provider: worker.agent,
|
|
244
|
+
...(worker.model ? { model: worker.model } : {}),
|
|
245
|
+
...(worker.reasoningEffort ? { reasoningEffort: worker.reasoningEffort } : {}),
|
|
246
|
+
}));
|
|
247
|
+
const parallelStories = parallel > 1 || candidates > 1 ? loadPrd(path).filter(story => !story.passes) : [];
|
|
248
|
+
const ambiguousAffinityProvider = [...new Set(parallelStories.flatMap(story => story.agent ? [story.agent] : []))]
|
|
249
|
+
.find(agent => parallelAffinityProviders.filter(provider => provider.provider === agent).length > 1);
|
|
250
|
+
if (ambiguousAffinityProvider) {
|
|
251
|
+
console.error(`Parallel affinity provider "${ambiguousAffinityProvider}" has multiple profiles. Configure exactly one profile for each story agent.`);
|
|
252
|
+
return 2;
|
|
253
|
+
}
|
|
254
|
+
if (opts.runner) {
|
|
255
|
+
const mismatchedAffinity = parallelStories.find(story => {
|
|
256
|
+
if (!story.agent)
|
|
257
|
+
return false;
|
|
258
|
+
const provider = parallelAffinityProviders.find(candidate => candidate.provider === story.agent)
|
|
259
|
+
?? parallelProviders.find(candidate => candidate.provider === story.agent);
|
|
260
|
+
return !provider
|
|
261
|
+
|| provider.provider !== runnerAgent
|
|
262
|
+
|| provider.model !== runnerSelection.model
|
|
263
|
+
|| provider.reasoningEffort !== runnerSelection.reasoningEffort;
|
|
264
|
+
});
|
|
265
|
+
if (mismatchedAffinity) {
|
|
266
|
+
console.error(`Injected runner cannot truthfully execute affinity provider for story ${mismatchedAffinity.id}. Remove the affinity or use the configured provider runner.`);
|
|
267
|
+
return 2;
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
const ambiguityPolicy = opts.decisionPolicy ?? opts.onAmbiguity ?? config.loop.decisionPolicy ?? config.loop.onAmbiguity ?? 'auto';
|
|
147
271
|
if (routingEnabled && (!config.routing || config.routing.workers.length === 0)) {
|
|
148
272
|
console.error('Adaptive routing was requested, but no worker profiles are configured. Run yoke setup . --routing or add routing.workers to .yoke/config.yaml.');
|
|
149
273
|
return 2;
|
|
150
274
|
}
|
|
275
|
+
const intake = opts.intake ?? (() => runChangeApply(targetDir, {
|
|
276
|
+
runner: runnerAgent,
|
|
277
|
+
reviewer: opts.reviewer ?? runnerAgent,
|
|
278
|
+
timeoutMs: idleMs,
|
|
279
|
+
isAvailable: available,
|
|
280
|
+
permissions,
|
|
281
|
+
selection: {
|
|
282
|
+
model: config.runner?.model,
|
|
283
|
+
reasoningEffort: config.runner?.reasoningEffort,
|
|
284
|
+
bare: config.runner?.bare,
|
|
285
|
+
},
|
|
286
|
+
commit: (_path, request) => commitPaths(targetDir, ['.yoke/prd.yaml'], `yoke: plan change ${request.id}`, commitIdentity),
|
|
287
|
+
}));
|
|
151
288
|
let runner = opts.runner;
|
|
152
289
|
if (!runner) {
|
|
153
|
-
|
|
154
|
-
|
|
290
|
+
const requiredProviders = parallel > 1 || candidates > 1
|
|
291
|
+
? [...new Set(loadPrd(path).filter(story => !story.passes).map(story => story.agent ?? runnerAgent))]
|
|
292
|
+
: [runnerAgent];
|
|
293
|
+
const unavailableProvider = requiredProviders.find(agent => !available(agent));
|
|
294
|
+
if (unavailableProvider) {
|
|
295
|
+
console.error(`Agent CLI "${unavailableProvider}" was not found on PATH. Install it, or pick another with --runner=<claude|codex|gemini>.`);
|
|
155
296
|
return 2;
|
|
156
297
|
}
|
|
157
298
|
// Token reporting is part of the machine interface: in --json mode a claude
|
|
158
299
|
// runner switches to stream-json so cumulative usage rides on every status.
|
|
159
300
|
const runnerOpts = {
|
|
160
301
|
tokenReport: opts.json === true,
|
|
161
|
-
onAmbiguity:
|
|
302
|
+
onAmbiguity: ambiguityPolicy,
|
|
162
303
|
perfCommand: config.perf?.command,
|
|
163
304
|
permissions,
|
|
164
|
-
selection:
|
|
165
|
-
model: config.runner?.model,
|
|
166
|
-
reasoningEffort: config.runner?.reasoningEffort,
|
|
167
|
-
bare: config.runner?.bare,
|
|
168
|
-
...((routingEnabled || opts.routing === false) ? { nativeMultiAgent: false } : {}),
|
|
169
|
-
},
|
|
305
|
+
selection: runnerSelection,
|
|
170
306
|
};
|
|
171
307
|
runner = routingEnabled && config.routing
|
|
172
308
|
? makeAdaptiveRunner({
|
|
@@ -220,8 +356,75 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
220
356
|
if (lock.stalePid !== undefined) {
|
|
221
357
|
console.warn(`Took over a stale loop lock (pid ${lock.stalePid} is gone).`);
|
|
222
358
|
}
|
|
359
|
+
const reporter = opts.reporter ?? makeReporter(targetDir, { json: opts.json });
|
|
360
|
+
const buildResume = (storyId, requestId) => buildTrustedDecisionResumeState({
|
|
361
|
+
storyId,
|
|
362
|
+
requestId,
|
|
363
|
+
...(opts.maxIterations !== undefined ? { maxIterations: opts.maxIterations } : {}),
|
|
364
|
+
agent: runnerAgent,
|
|
365
|
+
isolate: parallel > 1 || candidates > 1 || (opts.isolate ?? false),
|
|
366
|
+
reviewer: opts.reviewer,
|
|
367
|
+
review: opts.review === true || opts.reviewRunner !== undefined,
|
|
368
|
+
allowSelfReview: opts.allowSelfReview ?? false,
|
|
369
|
+
timeoutMinutes: opts.timeoutMinutes ?? config.loop.timeoutMinutes,
|
|
370
|
+
json: opts.json ?? false,
|
|
371
|
+
onAmbiguity: opts.decisionPolicy ? undefined : opts.onAmbiguity === 'resolve' || opts.onAmbiguity === 'abort' ? opts.onAmbiguity : (config.loop.decisionPolicy ? undefined : config.loop.onAmbiguity),
|
|
372
|
+
decisionPolicy: opts.decisionPolicy ?? (opts.onAmbiguity ? (opts.onAmbiguity === 'auto' || opts.onAmbiguity === 'critical' ? opts.onAmbiguity : undefined) : config.loop.decisionPolicy),
|
|
373
|
+
permissions,
|
|
374
|
+
parallel,
|
|
375
|
+
routing: routingEnabled,
|
|
376
|
+
...(qualityOverrides.quality !== undefined ? { quality: qualityOverrides.quality } : {}),
|
|
377
|
+
...(opts.qualityRounds !== undefined ? { qualityRounds: opts.qualityRounds } : {}),
|
|
378
|
+
...(opts.qualityMinutes !== undefined ? { qualityMinutes: opts.qualityMinutes } : {}),
|
|
379
|
+
...(opts.qualityPolicy ? { qualityPolicy: opts.qualityPolicy } : {}),
|
|
380
|
+
...(opts.candidates !== undefined ? { candidates: opts.candidates } : {}),
|
|
381
|
+
});
|
|
382
|
+
const reconcileDecisionResume = () => {
|
|
383
|
+
try {
|
|
384
|
+
const pendingDecision = readPendingDecision(targetDir);
|
|
385
|
+
if (pendingDecision) {
|
|
386
|
+
writeDecisionResume(targetDir, buildResume(pendingDecision.storyId, decisionRequestId(pendingDecision)));
|
|
387
|
+
}
|
|
388
|
+
else {
|
|
389
|
+
clearDecisionResume(targetDir);
|
|
390
|
+
}
|
|
391
|
+
return undefined;
|
|
392
|
+
}
|
|
393
|
+
catch (error) {
|
|
394
|
+
reporter.blocked(`could not persist trusted decision resume state: ${error instanceof Error ? error.message : String(error)}`);
|
|
395
|
+
return 1;
|
|
396
|
+
}
|
|
397
|
+
};
|
|
398
|
+
if (parallel > 1 || candidates > 1) {
|
|
399
|
+
return runParallelLoopCommand({
|
|
400
|
+
targetDir,
|
|
401
|
+
prdPath: path,
|
|
402
|
+
maxConcurrency: parallel,
|
|
403
|
+
candidateCount: candidates,
|
|
404
|
+
maxIterations: opts.maxIterations ?? Number.POSITIVE_INFINITY,
|
|
405
|
+
runner: opts.runner,
|
|
406
|
+
runnerAgent,
|
|
407
|
+
idleMs,
|
|
408
|
+
permissions,
|
|
409
|
+
selection: runnerSelection,
|
|
410
|
+
providers: parallelProviders,
|
|
411
|
+
affinityProviders: parallelAffinityProviders,
|
|
412
|
+
onAmbiguity: ambiguityPolicy,
|
|
413
|
+
git: opts.git,
|
|
414
|
+
identity: commitIdentity,
|
|
415
|
+
verify,
|
|
416
|
+
verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify)(dir),
|
|
417
|
+
requireCriterionEvidence: config.verify?.requireCriteria ?? false,
|
|
418
|
+
perf,
|
|
419
|
+
audit,
|
|
420
|
+
review,
|
|
421
|
+
reporter,
|
|
422
|
+
completion,
|
|
423
|
+
quality,
|
|
424
|
+
onCriticalDecision: decision => writeDecisionResume(targetDir, buildResume(decision.storyId, decisionRequestId(decision))),
|
|
425
|
+
}).then(code => reconcileDecisionResume() ?? code).finally(() => releaseLock(targetDir, lock.ownerToken));
|
|
426
|
+
}
|
|
223
427
|
try {
|
|
224
|
-
const reporter = opts.reporter ?? makeReporter(targetDir, { json: opts.json });
|
|
225
428
|
const maxIterations = opts.maxIterations ?? Number.POSITIVE_INFINITY;
|
|
226
429
|
const result = runLoop({
|
|
227
430
|
prdPath: path,
|
|
@@ -230,51 +433,22 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
230
433
|
git,
|
|
231
434
|
commitIdentity,
|
|
232
435
|
verify,
|
|
436
|
+
verifyCriterion: (dir, _story, criterion) => commandsVerifier(criterion.verify)(dir),
|
|
437
|
+
requireCriterionEvidence: config.verify?.requireCriteria ?? false,
|
|
438
|
+
completion,
|
|
439
|
+
intake,
|
|
233
440
|
perf,
|
|
234
441
|
audit,
|
|
235
442
|
maxIterations,
|
|
236
443
|
isolate: (opts.parallel ?? 1) > 1 ? true : (opts.isolate ?? false),
|
|
237
444
|
review,
|
|
238
445
|
reporter,
|
|
446
|
+
...(quality ?? {}),
|
|
447
|
+
...(quality ? { qualityEnabled: quality.qualityEnabled, qualityMetadata: quality.qualityMetadata } : {}),
|
|
239
448
|
});
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
writeDecisionResume(targetDir, {
|
|
244
|
-
version: 1,
|
|
245
|
-
storyId: pendingDecision.storyId,
|
|
246
|
-
requestId: decisionRequestId(pendingDecision),
|
|
247
|
-
answered: false,
|
|
248
|
-
...(opts.maxIterations !== undefined ? { maxIterations: opts.maxIterations } : {}),
|
|
249
|
-
agent: runnerAgent,
|
|
250
|
-
isolate: opts.isolate ?? false,
|
|
251
|
-
reviewer: opts.reviewer,
|
|
252
|
-
review: opts.review === true || opts.reviewRunner !== undefined,
|
|
253
|
-
allowSelfReview: opts.allowSelfReview ?? false,
|
|
254
|
-
timeoutMinutes: opts.timeoutMinutes ?? config.loop.timeoutMinutes,
|
|
255
|
-
json: opts.json ?? false,
|
|
256
|
-
onAmbiguity: opts.decisionPolicy
|
|
257
|
-
? undefined
|
|
258
|
-
: opts.onAmbiguity === 'resolve' || opts.onAmbiguity === 'abort'
|
|
259
|
-
? opts.onAmbiguity
|
|
260
|
-
: (config.loop.decisionPolicy ? undefined : config.loop.onAmbiguity),
|
|
261
|
-
decisionPolicy: opts.decisionPolicy
|
|
262
|
-
?? (opts.onAmbiguity
|
|
263
|
-
? (opts.onAmbiguity === 'auto' || opts.onAmbiguity === 'critical' ? opts.onAmbiguity : undefined)
|
|
264
|
-
: config.loop.decisionPolicy),
|
|
265
|
-
permissions,
|
|
266
|
-
parallel: opts.parallel ?? 1,
|
|
267
|
-
routing: routingEnabled,
|
|
268
|
-
});
|
|
269
|
-
}
|
|
270
|
-
else
|
|
271
|
-
clearDecisionResume(targetDir);
|
|
272
|
-
}
|
|
273
|
-
catch (error) {
|
|
274
|
-
const reason = `could not persist trusted decision resume state: ${error.message}`;
|
|
275
|
-
reporter.blocked(reason);
|
|
276
|
-
return 1;
|
|
277
|
-
}
|
|
449
|
+
const resumeCode = reconcileDecisionResume();
|
|
450
|
+
if (resumeCode !== undefined)
|
|
451
|
+
return resumeCode;
|
|
278
452
|
// In json mode stdout belongs to the NDJSON stream — route the narrative summary to stderr.
|
|
279
453
|
const say = opts.json ? (line) => console.error(line) : (line) => console.log(line);
|
|
280
454
|
say(`Loop ${result.status} after ${result.iterations} iteration(s): ${result.finalProgress.passed}/${result.finalProgress.total} stories pass`);
|
package/dist/loop/runner.js
CHANGED
|
@@ -1,16 +1,25 @@
|
|
|
1
|
+
import { isAcceptanceCriterion } from './prd.js';
|
|
1
2
|
import { execFileSync, execSync } from 'node:child_process';
|
|
2
|
-
import { existsSync
|
|
3
|
+
import { existsSync } from 'node:fs';
|
|
4
|
+
import { createRequire } from 'node:module';
|
|
3
5
|
import { join } from 'node:path';
|
|
4
|
-
import { fileURLToPath } from 'node:url';
|
|
6
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
5
7
|
import { loadContext, formatForPrompt, contextDir } from '../context/context.js';
|
|
6
|
-
import { buildProviderInvocation } from '../agents/providers.js';
|
|
7
|
-
import { parseProviderTelemetry } from '../agents/telemetry.js';
|
|
8
|
-
import { formatReviewContract,
|
|
8
|
+
import { buildProviderInvocation, startProviderProcess } from '../agents/providers.js';
|
|
9
|
+
import { parseProviderResult, parseProviderTelemetry } from '../agents/telemetry.js';
|
|
10
|
+
import { formatReviewContract, formatReviewStdoutContract, parseReviewVerdict } from '../review/verdict.js';
|
|
9
11
|
export function contextBlockFor(targetDir) {
|
|
10
12
|
return formatForPrompt(loadContext(contextDir(targetDir)));
|
|
11
13
|
}
|
|
14
|
+
function formatAcceptance(story) {
|
|
15
|
+
return story.acceptance.map(criterion => {
|
|
16
|
+
if (!isAcceptanceCriterion(criterion))
|
|
17
|
+
return `- ${criterion}`;
|
|
18
|
+
return `- [${criterion.id}] ${criterion.text}\n Proof: ${criterion.verify.join(' && ')}`;
|
|
19
|
+
}).join('\n');
|
|
20
|
+
}
|
|
12
21
|
export function buildClaudePrompt(story, context, onAmbiguity = 'resolve', perfCommand) {
|
|
13
|
-
const criteria = story
|
|
22
|
+
const criteria = formatAcceptance(story);
|
|
14
23
|
const lines = [
|
|
15
24
|
'You are an autonomous coding agent running inside the Yoke loop.',
|
|
16
25
|
'Implement ONLY this story and nothing else. Follow test-driven development.',
|
|
@@ -32,8 +41,8 @@ export function buildClaudePrompt(story, context, onAmbiguity = 'resolve', perfC
|
|
|
32
41
|
lines.push('- Keep your final message to a few short sentences: what changed and what you verified.');
|
|
33
42
|
return lines.join('\n');
|
|
34
43
|
}
|
|
35
|
-
export function buildReviewPrompt(story, context, verdictPath) {
|
|
36
|
-
const criteria = story
|
|
44
|
+
export function buildReviewPrompt(story, context, verdictPath, provider) {
|
|
45
|
+
const criteria = formatAcceptance(story);
|
|
37
46
|
const lines = [
|
|
38
47
|
'You are an independent reviewer inside the Yoke loop. You did NOT implement this change.',
|
|
39
48
|
'Review the current uncommitted working-tree changes against the story below.',
|
|
@@ -43,11 +52,10 @@ export function buildReviewPrompt(story, context, verdictPath) {
|
|
|
43
52
|
lines.push('', `Story ${story.id}: ${story.title}`, 'Acceptance criteria:', criteria, '', 'Approve ONLY if every acceptance criterion is met and the change is sound.', 'If you find ANY blocking issue (an unmet criterion, a bug, a missing test), reject.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', verdictPath
|
|
44
53
|
? 'Do not modify project source, tests, configuration, or generated artifacts. The verdict file named below is the only permitted write. Do not commit.'
|
|
45
54
|
: 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
|
|
46
|
-
|
|
47
|
-
lines.push('', formatReviewContract(verdictPath));
|
|
55
|
+
lines.push('', verdictPath ? formatReviewContract(verdictPath, provider) : formatReviewStdoutContract(provider ?? 'claude'));
|
|
48
56
|
return lines.join('\n');
|
|
49
57
|
}
|
|
50
|
-
export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
|
|
58
|
+
export function buildStandaloneReviewPrompt(scope, focus, verdictPath, provider) {
|
|
51
59
|
const lines = [
|
|
52
60
|
'You are an independent reviewer. You did NOT write this change.',
|
|
53
61
|
`Review ${scope}. Run git yourself to see the diff (e.g. \`git diff\`, or \`git diff <base>..HEAD\`).`,
|
|
@@ -58,8 +66,7 @@ export function buildStandaloneReviewPrompt(scope, focus, verdictPath) {
|
|
|
58
66
|
lines.push('', 'Approve by exiting 0 ONLY if the change is sound and complete.', 'If you find ANY blocking issue, exit non-zero to reject and explain what is wrong.', 'Base your verdict only on what the diff and test runs actually show — never assume unverified behavior.', verdictPath
|
|
59
67
|
? 'Do not modify project source, tests, configuration, or generated artifacts. The verdict file named below is the only permitted write. Do not commit.'
|
|
60
68
|
: 'Do not modify files. Do not commit.', 'Keep your verdict to a few short sentences.');
|
|
61
|
-
|
|
62
|
-
lines.push('', formatReviewContract(verdictPath));
|
|
69
|
+
lines.push('', verdictPath ? formatReviewContract(verdictPath, provider) : formatReviewStdoutContract(provider ?? 'claude'));
|
|
63
70
|
return lines.join('\n');
|
|
64
71
|
}
|
|
65
72
|
// Headless agents must run non-interactively: with plain `-p` the CLI denies
|
|
@@ -142,9 +149,13 @@ export function parseClaudeStreamUsage(lines) {
|
|
|
142
149
|
const usage = result ?? { inputTokens: assistantIn, outputTokens: assistantOut };
|
|
143
150
|
return model ? { ...usage, model } : usage;
|
|
144
151
|
}
|
|
145
|
-
function
|
|
146
|
-
|
|
147
|
-
|
|
152
|
+
function watchdogArgs() {
|
|
153
|
+
const compiled = fileURLToPath(new URL('./watchdog.js', import.meta.url));
|
|
154
|
+
if (existsSync(compiled))
|
|
155
|
+
return [compiled];
|
|
156
|
+
const source = fileURLToPath(new URL('./watchdog.ts', import.meta.url));
|
|
157
|
+
const tsxLoader = pathToFileURL(createRequire(import.meta.url).resolve('tsx')).href;
|
|
158
|
+
return ['--import', tsxLoader, source];
|
|
148
159
|
}
|
|
149
160
|
// When idleTimeoutMs > 0, run the agent THROUGH the watchdog so a silent hang is
|
|
150
161
|
// killed after idleTimeoutMs of no output. The prompt still flows via stdin.
|
|
@@ -153,14 +164,14 @@ function watchdogPath() {
|
|
|
153
164
|
// killing by process-name/command-line pattern takes down other projects'
|
|
154
165
|
// runners too. (Plain repos, e.g. `yoke review` outside a yoke project, get
|
|
155
166
|
// no pid file rather than a littered .yoke dir.)
|
|
156
|
-
export function buildWatchdogInvocation(inv, idleTimeoutMs) {
|
|
167
|
+
export function buildWatchdogInvocation(inv, idleTimeoutMs, ownershipRoot = inv.cwd) {
|
|
157
168
|
if (idleTimeoutMs <= 0)
|
|
158
169
|
return inv;
|
|
159
|
-
const yokeDir = join(
|
|
170
|
+
const yokeDir = join(ownershipRoot, '.yoke');
|
|
160
171
|
const pidArgs = existsSync(yokeDir) ? [`--pid-file=${join(yokeDir, 'runner.pid')}`] : [];
|
|
161
172
|
return {
|
|
162
173
|
command: 'node',
|
|
163
|
-
args: [
|
|
174
|
+
args: [...watchdogArgs(), `--idle-ms=${idleTimeoutMs}`, ...pidArgs, '--', inv.command, ...inv.args],
|
|
164
175
|
input: inv.input,
|
|
165
176
|
cwd: inv.cwd,
|
|
166
177
|
};
|
|
@@ -179,7 +190,7 @@ export function win32CommandString(command, args) {
|
|
|
179
190
|
return [command, ...args].map(q).join(' ');
|
|
180
191
|
}
|
|
181
192
|
function runCli(inv) {
|
|
182
|
-
if (process.platform === 'win32') {
|
|
193
|
+
if (process.platform === 'win32' && !/\.(?:exe|com)$/iu.test(inv.command) && inv.command !== process.execPath && inv.command !== 'node') {
|
|
183
194
|
execSync(win32CommandString(inv.command, inv.args), {
|
|
184
195
|
cwd: inv.cwd,
|
|
185
196
|
input: inv.input,
|
|
@@ -200,7 +211,7 @@ function runCli(inv) {
|
|
|
200
211
|
// through it. Throws on a non-zero exit; the error carries the partial stdout.
|
|
201
212
|
function runCliCapture(inv) {
|
|
202
213
|
const opts = { cwd: inv.cwd, input: inv.input, stdio: ['pipe', 'pipe', 'inherit'], encoding: 'utf8', maxBuffer: 64 * 1024 * 1024 };
|
|
203
|
-
return process.platform === 'win32'
|
|
214
|
+
return process.platform === 'win32' && !/\.(?:exe|com)$/iu.test(inv.command) && inv.command !== process.execPath && inv.command !== 'node'
|
|
204
215
|
? execSync(win32CommandString(inv.command, inv.args), opts)
|
|
205
216
|
: execFileSync(inv.command, inv.args, opts);
|
|
206
217
|
}
|
|
@@ -217,7 +228,7 @@ function runReviewCli(inv) {
|
|
|
217
228
|
encoding: 'utf8',
|
|
218
229
|
maxBuffer: 64 * 1024 * 1024,
|
|
219
230
|
};
|
|
220
|
-
if (process.platform === 'win32')
|
|
231
|
+
if (process.platform === 'win32' && !/\.(?:exe|com)$/iu.test(inv.command) && inv.command !== process.execPath && inv.command !== 'node')
|
|
221
232
|
execSync(win32CommandString(inv.command, inv.args), opts);
|
|
222
233
|
else
|
|
223
234
|
execFileSync(inv.command, inv.args, opts);
|
|
@@ -288,6 +299,9 @@ export function runReviewAgent(inv) {
|
|
|
288
299
|
return { success: false, summary: processFailureSummary(error) };
|
|
289
300
|
}
|
|
290
301
|
}
|
|
302
|
+
export function makeAsyncRunner(agent, opts = {}) {
|
|
303
|
+
return (ctx) => startProviderProcess(agent, runnerInvocation(agent, buildClaudePrompt(ctx.story, contextBlockFor(ctx.targetDir), opts.onAmbiguity, opts.perfCommand), ctx.targetDir, true, opts.permissions ?? 'safe', opts.selection), opts.process);
|
|
304
|
+
}
|
|
291
305
|
export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
|
|
292
306
|
// Claude always streams (see runnerInvocation) — capture the stream so tokens are
|
|
293
307
|
// always reported; other agents keep inherit stdio. opts.tokenReport is now
|
|
@@ -323,33 +337,62 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
|
|
|
323
337
|
};
|
|
324
338
|
}
|
|
325
339
|
export const claudeRunner = makeRunner('claude');
|
|
326
|
-
export function makeReviewRunner(agent, idleTimeoutMs = 0, exec
|
|
340
|
+
export function makeReviewRunner(agent, idleTimeoutMs = 0, exec) {
|
|
327
341
|
return (ctx) => {
|
|
328
|
-
const
|
|
329
|
-
|
|
330
|
-
rmSync(verdictPath, { force: true });
|
|
331
|
-
const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir), verdictPath), ctx.targetDir, 'safe');
|
|
342
|
+
const before = repositoryFingerprint(ctx.targetDir);
|
|
343
|
+
const base = agentInvocation(agent, buildReviewPrompt(ctx.story, contextBlockFor(ctx.targetDir), undefined, agent), ctx.targetDir, 'read-only');
|
|
332
344
|
const inv = buildWatchdogInvocation(base, idleTimeoutMs);
|
|
333
345
|
let processFailure;
|
|
346
|
+
let actualModel;
|
|
347
|
+
let output = '';
|
|
334
348
|
try {
|
|
335
|
-
exec(inv);
|
|
349
|
+
const result = exec?.(inv) ?? runCapturedAgent(agent, inv);
|
|
350
|
+
if (!result.success)
|
|
351
|
+
processFailure = result.summary;
|
|
352
|
+
actualModel = result.tokens?.model;
|
|
353
|
+
output = result.output;
|
|
354
|
+
if (!exec && !actualModel && !processFailure)
|
|
355
|
+
processFailure = 'review provider did not report its model';
|
|
336
356
|
}
|
|
337
357
|
catch (e) {
|
|
338
358
|
processFailure = processFailureSummary(e);
|
|
339
359
|
}
|
|
340
360
|
try {
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
361
|
+
if (repositoryFingerprint(ctx.targetDir) !== before)
|
|
362
|
+
throw new Error('reviewer modified the repository during a read-only review');
|
|
363
|
+
const expected = { provider: agent, ...(actualModel ? { model: actualModel } : {}) };
|
|
364
|
+
const verdict = parseReviewVerdict(parseProviderResult(agent, output), expected);
|
|
365
|
+
if (processFailure) {
|
|
366
|
+
return {
|
|
367
|
+
success: false,
|
|
368
|
+
summary: `review process failed: ${processFailure}; verdict: ${verdict.summary}`,
|
|
369
|
+
reviewOutcome: { kind: 'infrastructure', summary: processFailure },
|
|
370
|
+
};
|
|
371
|
+
}
|
|
344
372
|
return verdict.approved
|
|
345
|
-
?
|
|
346
|
-
:
|
|
373
|
+
? reviewResult(agent, ctx.story.id, verdict, { kind: 'approved', verdict })
|
|
374
|
+
: reviewResult(agent, ctx.story.id, verdict, { kind: 'rejected', verdict });
|
|
347
375
|
}
|
|
348
376
|
catch (e) {
|
|
349
|
-
|
|
377
|
+
const summary = `${processFailure ? `review process failed: ${processFailure}; ` : ''}${e.message}`;
|
|
378
|
+
return { success: false, summary, reviewOutcome: processFailure ? { kind: 'infrastructure', summary } : { kind: 'malformed', summary } };
|
|
350
379
|
}
|
|
351
380
|
};
|
|
352
381
|
}
|
|
382
|
+
export function repositoryFingerprint(targetDir) {
|
|
383
|
+
try {
|
|
384
|
+
return execFileSync('git', ['diff', '--binary', 'HEAD'], { cwd: targetDir, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] })
|
|
385
|
+
+ execFileSync('git', ['status', '--porcelain=v1', '-z'], { cwd: targetDir, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] });
|
|
386
|
+
}
|
|
387
|
+
catch {
|
|
388
|
+
return '';
|
|
389
|
+
}
|
|
390
|
+
}
|
|
391
|
+
function reviewResult(agent, storyId, verdict, reviewOutcome) {
|
|
392
|
+
return verdict.approved
|
|
393
|
+
? { success: true, summary: `${agent} approved ${storyId}: ${verdict.summary}`, reviewOutcome }
|
|
394
|
+
: { success: false, summary: `${agent} rejected ${storyId}: ${verdict.summary}`, reviewOutcome };
|
|
395
|
+
}
|
|
353
396
|
// Probe whether the agent's CLI is on PATH (so the loop can refuse upfront with a
|
|
354
397
|
// clear message instead of failing mid-run with spawn ENOENT). Never throws.
|
|
355
398
|
export function isAgentAvailable(agent) {
|
package/dist/loop/verify.js
CHANGED
|
@@ -16,6 +16,17 @@ export function commandVerifier(command) {
|
|
|
16
16
|
}
|
|
17
17
|
};
|
|
18
18
|
}
|
|
19
|
+
/** Execute the proof commands attached to one acceptance criterion. */
|
|
20
|
+
export function commandsVerifier(commands) {
|
|
21
|
+
return (targetDir) => {
|
|
22
|
+
for (const command of commands) {
|
|
23
|
+
const result = commandVerifier(command)(targetDir);
|
|
24
|
+
if (!result.passed)
|
|
25
|
+
return result;
|
|
26
|
+
}
|
|
27
|
+
return { passed: true, summary: `${commands.length} criterion command${commands.length === 1 ? '' : 's'} passed` };
|
|
28
|
+
};
|
|
29
|
+
}
|
|
19
30
|
// Re-run a failing verifier up to `retries` times; the first pass wins. Lets a
|
|
20
31
|
// transient flake (e.g. a load-induced async timeout) self-heal while a real
|
|
21
32
|
// failure still fails (it stays red across every attempt).
|