@hecer/yoke 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +22 -0
  4. package/README.md +67 -15
  5. package/TODOS.md +0 -3
  6. package/canon/loop/loop-spec.md +22 -8
  7. package/canon/manifest.yaml +1 -1
  8. package/dist/agents/contracts.js +50 -0
  9. package/dist/agents/process-incarnation.js +15 -0
  10. package/dist/agents/process-record.js +65 -0
  11. package/dist/agents/process-streams.js +40 -0
  12. package/dist/agents/process.js +177 -0
  13. package/dist/agents/providers.js +10 -7
  14. package/dist/agents/telemetry.js +62 -0
  15. package/dist/cli.js +55 -3
  16. package/dist/loop/candidate-boundaries.js +43 -0
  17. package/dist/loop/candidate-cleanup.js +98 -0
  18. package/dist/loop/candidate-contracts.js +1 -0
  19. package/dist/loop/candidate-selection.js +84 -0
  20. package/dist/loop/candidates.js +228 -0
  21. package/dist/loop/claim-lease.js +131 -0
  22. package/dist/loop/claims.js +177 -40
  23. package/dist/loop/cleanup.js +117 -15
  24. package/dist/loop/decision.js +31 -0
  25. package/dist/loop/dispatcher.js +334 -0
  26. package/dist/loop/loop.js +109 -16
  27. package/dist/loop/merge-queue.js +12 -6
  28. package/dist/loop/parallel-adapters.js +185 -0
  29. package/dist/loop/parallel-command.js +287 -0
  30. package/dist/loop/parallel.js +2 -4
  31. package/dist/loop/prd.js +4 -1
  32. package/dist/loop/reporter.js +86 -5
  33. package/dist/loop/run-command.js +204 -51
  34. package/dist/loop/runner.js +67 -32
  35. package/dist/loop/watchdog.js +67 -8
  36. package/dist/loop/worker-cancellation.js +17 -0
  37. package/dist/loop/worker-cleanup.js +23 -0
  38. package/dist/loop/worker-contracts.js +1 -0
  39. package/dist/loop/worker.js +254 -0
  40. package/dist/quality/artifacts.js +59 -0
  41. package/dist/quality/candidate-comparison.js +130 -0
  42. package/dist/quality/command.js +316 -0
  43. package/dist/quality/loop.js +86 -0
  44. package/dist/quality/process-command.js +57 -0
  45. package/dist/quality/reference.js +187 -0
  46. package/dist/quality/repair.js +11 -0
  47. package/dist/quality/runner.js +66 -0
  48. package/dist/quality/types.js +60 -0
  49. package/dist/quality/verdict.js +142 -0
  50. package/dist/retrofit/config.js +4 -2
  51. package/dist/retrofit/gitignore.js +3 -0
  52. package/dist/review/command.js +27 -38
  53. package/dist/review/verdict.js +38 -7
  54. package/docs/MIGRATING-TO-1.4.md +70 -0
  55. package/docs/PUBLISHING.md +16 -2
  56. package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -0
  57. package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -0
  58. package/gemini-extension.json +1 -1
  59. package/package.json +1 -1
@@ -0,0 +1,66 @@
1
+ import { join } from 'node:path';
2
+ import { createHash } from 'node:crypto';
3
+ import { parseProviderResult } from '../agents/telemetry.js';
4
+ import { storyPathSegment } from '../loop/prd.js';
5
+ import { reduceComparisons } from './verdict.js';
6
+ export function runQualityCritic(input) {
7
+ const segment = input.evidenceScope ? `${input.evidenceScope}-round-${input.round}` : `round-${input.round}`;
8
+ const root = join(input.targetDir, '.yoke', 'proof', storyPathSegment(input.storyId), 'quality', segment);
9
+ input.mkdir(root);
10
+ const first = input.firstLabel();
11
+ const second = first === 'A' ? 'B' : 'A';
12
+ const attemptPrefix = input.attemptIdPrefix ?? String(input.round);
13
+ const normalRequest = request(input, `${attemptPrefix}-normal`, first);
14
+ const swappedRequest = request(input, `${attemptPrefix}-swapped`, second);
15
+ const normal = input.invoke(structuredClone(normalRequest));
16
+ if (!normal.ok)
17
+ return { kind: 'infrastructure', summary: normal.summary };
18
+ const swapped = input.invoke(structuredClone(swappedRequest));
19
+ if (!swapped.ok)
20
+ return { kind: 'infrastructure', summary: swapped.summary };
21
+ if (normal.actualModel && swapped.actualModel && normal.actualModel !== swapped.actualModel)
22
+ return { kind: 'infrastructure', summary: `critic model changed between comparisons: ${normal.actualModel} -> ${swapped.actualModel}` };
23
+ const actualModel = normal.actualModel ?? swapped.actualModel ?? input.model;
24
+ if (!actualModel)
25
+ return { kind: 'infrastructure', summary: 'critic did not report its model and no explicit model was configured' };
26
+ const expected = {
27
+ normal: expectation(normalRequest),
28
+ swapped: expectation(swappedRequest),
29
+ provenance: { provider: input.provider, model: actualModel },
30
+ };
31
+ input.writeFile(join(root, 'normal.verdict.json'), JSON.stringify({ provider: input.provider, model: actualModel, raw: normal.output }));
32
+ input.writeFile(join(root, 'swapped.verdict.json'), JSON.stringify({ provider: input.provider, model: actualModel, raw: swapped.output }));
33
+ const reduction = reduceComparisons({
34
+ expected,
35
+ normal: parseProviderResult(input.provider, normal.output),
36
+ swapped: parseProviderResult(input.provider, swapped.output),
37
+ });
38
+ return input.policy === 'advisory' ? { kind: 'skipped', summary: 'advisory comparison recorded' } : reduction;
39
+ }
40
+ function request(input, attemptId, candidateLabel) {
41
+ const issued = {
42
+ permissions: 'read-only',
43
+ attemptId,
44
+ candidateLabel,
45
+ referenceLabel: candidateLabel === 'A' ? 'B' : 'A',
46
+ trustedRubric: input.rubric,
47
+ reference: { ...input.reference, artifact: input.reference.artifact ?? 'reference.bin' },
48
+ candidate: {
49
+ digests: [...input.candidate.digests],
50
+ artifacts: input.candidate.artifacts ? [...input.candidate.artifacts] : input.candidate.digests.map((_, index) => `candidate-${index + 1}.bin`),
51
+ },
52
+ };
53
+ return { ...issued, promptDigest: digest(JSON.stringify(issued)), rubricDigest: digest(input.rubric) };
54
+ }
55
+ function expectation(request) {
56
+ return {
57
+ attemptId: request.attemptId,
58
+ candidateDigests: [...request.candidate.digests],
59
+ referenceDigest: request.reference.digest,
60
+ promptDigest: request.promptDigest,
61
+ rubricDigest: request.rubricDigest,
62
+ };
63
+ }
64
+ function digest(value) {
65
+ return createHash('sha256').update(value).digest('hex');
66
+ }
@@ -0,0 +1,60 @@
1
+ import { z } from 'zod';
2
+ const QualityPolicySchema = z.enum(['blocking', 'advisory']);
3
+ const QualityAgentSchema = z.enum(['claude', 'codex', 'gemini']);
4
+ const RelativePathSchema = z.string().min(1).refine(value => !/^(?:[A-Za-z]:[\\/]|[\\/])/.test(value) && !value.split(/[\\/]+/).includes('..'), 'path must stay within the project');
5
+ const QualityReferenceSchema = z.object({
6
+ name: z.string().min(1),
7
+ source: z.string().min(1),
8
+ kind: z.enum(['url', 'file', 'command']),
9
+ digest: z.string().min(1).optional(),
10
+ }).superRefine((value, context) => {
11
+ if (value.kind === 'file' && !RelativePathSchema.safeParse(value.source).success)
12
+ context.addIssue({ code: 'custom', path: ['source'], message: 'file reference must stay within the project' });
13
+ });
14
+ const QualityCandidateSchema = z.union([
15
+ z.object({ kind: z.enum(['screenshots', 'files']), paths: z.array(RelativePathSchema).min(1) }),
16
+ z.object({ kind: z.enum(['command-output', 'benchmark']), command: z.string().min(1) }),
17
+ ]);
18
+ export const ProjectQualityDefaultsSchema = z.object({
19
+ enabled: z.boolean().default(false),
20
+ policy: QualityPolicySchema.default('blocking'),
21
+ maxRounds: z.number().int().positive().default(3),
22
+ maxMinutes: z.number().int().positive().default(60),
23
+ consistencyChecks: z.literal(2).default(2),
24
+ maxParallelCandidates: z.number().int().positive().default(2),
25
+ criticAgent: QualityAgentSchema.optional(),
26
+ criticModel: z.string().min(1).optional(),
27
+ criticReasoningEffort: z.string().min(1).optional(),
28
+ repairAgent: QualityAgentSchema.optional(),
29
+ repairModel: z.string().min(1).optional(),
30
+ repairReasoningEffort: z.string().min(1).optional(),
31
+ critic: z.object({
32
+ agent: QualityAgentSchema.optional(),
33
+ model: z.string().min(1).optional(),
34
+ reasoningEffort: z.string().min(1).optional(),
35
+ }).optional(),
36
+ repair: z.object({
37
+ agent: QualityAgentSchema.optional(),
38
+ model: z.string().min(1).optional(),
39
+ reasoningEffort: z.string().min(1).optional(),
40
+ }).optional(),
41
+ });
42
+ export const StoryQualityDeclarationSchema = z.object({
43
+ reference: QualityReferenceSchema,
44
+ candidate: QualityCandidateSchema,
45
+ rubric: z.string().min(1),
46
+ policy: QualityPolicySchema.optional(),
47
+ });
48
+ export function resolveQualityPolicy(input) {
49
+ const overrides = input.overrides;
50
+ const unbounded = overrides?.qualityUnbounded === true;
51
+ const enabled = input.declaration !== undefined && (unbounded || (overrides?.quality ?? input.defaults?.enabled ?? false));
52
+ const policy = overrides?.qualityPolicy ?? input.declaration?.policy ?? input.defaults?.policy ?? 'blocking';
53
+ const maxRounds = overrides?.qualityRounds ?? input.defaults?.maxRounds ?? 3;
54
+ const maxMinutes = overrides?.qualityMinutes ?? input.defaults?.maxMinutes ?? 60;
55
+ return {
56
+ enabled,
57
+ policy,
58
+ limits: unbounded ? { unbounded: true } : { maxRounds, maxMinutes },
59
+ };
60
+ }
@@ -0,0 +1,142 @@
1
+ import { z } from 'zod';
2
+ export const QualityLabelSchema = z.enum(['A', 'B']);
3
+ const DigestSchema = z.string().min(1);
4
+ const ComparedArtifactSchema = z.object({ label: QualityLabelSchema, digest: DigestSchema });
5
+ const ProvenanceSchema = z.object({
6
+ provider: z.string().min(1),
7
+ model: z.string().min(1),
8
+ promptDigest: DigestSchema,
9
+ rubricDigest: DigestSchema,
10
+ referenceDigest: DigestSchema,
11
+ candidateDigest: DigestSchema,
12
+ });
13
+ export const QualityVerdictSchema = z.object({
14
+ schemaVersion: z.literal(1),
15
+ attemptId: z.string().min(1),
16
+ winner: z.enum(['candidate', 'reference']),
17
+ biggestGap: z.string().min(1),
18
+ evidence: z.array(z.string().min(1)).min(1),
19
+ confidence: z.enum(['high', 'medium', 'low']),
20
+ candidate: ComparedArtifactSchema,
21
+ reference: ComparedArtifactSchema,
22
+ provenance: ProvenanceSchema,
23
+ }).superRefine((value, context) => {
24
+ if (value.candidate.label === value.reference.label) {
25
+ context.addIssue({ code: z.ZodIssueCode.custom, message: 'candidate and reference labels must differ' });
26
+ }
27
+ if (value.candidate.digest !== value.provenance.candidateDigest || value.reference.digest !== value.provenance.referenceDigest) {
28
+ context.addIssue({ code: z.ZodIssueCode.custom, message: 'artifact digests must match provenance' });
29
+ }
30
+ });
31
+ const CandidatePairVerdictSchema = z.object({
32
+ schemaVersion: z.literal(1),
33
+ attemptId: z.string().min(1),
34
+ winner: QualityLabelSchema,
35
+ evidence: z.array(z.string().min(1)).min(1),
36
+ confidence: z.enum(['high', 'medium', 'low']),
37
+ left: ComparedArtifactSchema,
38
+ right: ComparedArtifactSchema,
39
+ provenance: z.object({
40
+ leftDigest: DigestSchema,
41
+ rightDigest: DigestSchema,
42
+ provider: z.string().min(1),
43
+ model: z.string().min(1),
44
+ promptDigest: DigestSchema,
45
+ rubricDigest: DigestSchema,
46
+ }),
47
+ }).superRefine((value, context) => {
48
+ if (value.left.label === value.right.label) {
49
+ context.addIssue({ code: z.ZodIssueCode.custom, message: 'candidate labels must differ' });
50
+ }
51
+ if (value.left.digest !== value.provenance.leftDigest || value.right.digest !== value.provenance.rightDigest) {
52
+ context.addIssue({ code: z.ZodIssueCode.custom, message: 'candidate digests must match provenance' });
53
+ }
54
+ });
55
+ export function assignBlindLabels(selectCandidateLabel) {
56
+ const candidate = selectCandidateLabel();
57
+ return { candidate, reference: candidate === 'A' ? 'B' : 'A' };
58
+ }
59
+ export function reduceComparisons(input) {
60
+ const normal = QualityVerdictSchema.safeParse(input.normal);
61
+ const swapped = QualityVerdictSchema.safeParse(input.swapped);
62
+ if (!normal.success || !swapped.success)
63
+ return { kind: 'inconsistent', reason: 'invalid-verdict' };
64
+ if (normal.data.attemptId === swapped.data.attemptId)
65
+ return { kind: 'inconsistent', reason: 'not-fresh' };
66
+ if (normal.data.attemptId !== input.expected.normal.attemptId || swapped.data.attemptId !== input.expected.swapped.attemptId) {
67
+ return { kind: 'inconsistent', reason: 'attempt-mismatch' };
68
+ }
69
+ if (normal.data.candidate.label === swapped.data.candidate.label || normal.data.reference.label === swapped.data.reference.label) {
70
+ return { kind: 'inconsistent', reason: 'labels-not-swapped' };
71
+ }
72
+ if (normal.data.candidate.digest !== swapped.data.candidate.digest || normal.data.reference.digest !== swapped.data.reference.digest) {
73
+ return { kind: 'inconsistent', reason: 'digest-mismatch' };
74
+ }
75
+ if (!input.expected.normal.candidateDigests.includes(normal.data.candidate.digest)
76
+ || !input.expected.swapped.candidateDigests.includes(swapped.data.candidate.digest)
77
+ || normal.data.reference.digest !== input.expected.normal.referenceDigest
78
+ || swapped.data.reference.digest !== input.expected.swapped.referenceDigest) {
79
+ return { kind: 'inconsistent', reason: 'digest-mismatch' };
80
+ }
81
+ if (normal.data.provenance.provider !== input.expected.provenance.provider
82
+ || swapped.data.provenance.provider !== input.expected.provenance.provider
83
+ || normal.data.provenance.model !== input.expected.provenance.model
84
+ || swapped.data.provenance.model !== input.expected.provenance.model
85
+ || normal.data.provenance.promptDigest !== input.expected.normal.promptDigest
86
+ || swapped.data.provenance.promptDigest !== input.expected.swapped.promptDigest
87
+ || normal.data.provenance.rubricDigest !== input.expected.normal.rubricDigest
88
+ || swapped.data.provenance.rubricDigest !== input.expected.swapped.rubricDigest) {
89
+ return { kind: 'inconsistent', reason: 'provenance-mismatch' };
90
+ }
91
+ if (normal.data.winner !== swapped.data.winner)
92
+ return { kind: 'inconsistent', reason: 'winner-disagrees' };
93
+ if (normal.data.winner === 'reference') {
94
+ return { kind: 'lose', reason: 'reference-selected', biggestGap: normal.data.biggestGap, evidence: normal.data.evidence };
95
+ }
96
+ if (normal.data.confidence === 'low' || swapped.data.confidence === 'low') {
97
+ return { kind: 'lose', reason: 'low-confidence', biggestGap: normal.data.biggestGap, evidence: normal.data.evidence };
98
+ }
99
+ return {
100
+ kind: 'pass',
101
+ candidateDigest: normal.data.candidate.digest,
102
+ referenceDigest: normal.data.reference.digest,
103
+ };
104
+ }
105
+ export function reduceCandidatePairComparisons(input) {
106
+ const normal = CandidatePairVerdictSchema.safeParse(input.normal);
107
+ const swapped = CandidatePairVerdictSchema.safeParse(input.swapped);
108
+ if (!normal.success || !swapped.success)
109
+ return { kind: 'inconsistent', reason: 'invalid-verdict' };
110
+ const normalMismatch = candidatePairMismatch(normal.data, input.expected.normal);
111
+ if (normalMismatch)
112
+ return { kind: 'inconsistent', reason: normalMismatch };
113
+ const swappedMismatch = candidatePairMismatch(swapped.data, input.expected.swapped);
114
+ if (swappedMismatch)
115
+ return { kind: 'inconsistent', reason: swappedMismatch };
116
+ if (normal.data.confidence === 'low' || swapped.data.confidence === 'low') {
117
+ return { kind: 'inconsistent', reason: 'low-confidence' };
118
+ }
119
+ const normalWinner = normal.data.winner === normal.data.left.label
120
+ ? { candidateId: input.expected.normal.candidateIds.left, digest: input.expected.normal.request.left.digest }
121
+ : { candidateId: input.expected.normal.candidateIds.right, digest: input.expected.normal.request.right.digest };
122
+ const swappedWinner = swapped.data.winner === swapped.data.left.label
123
+ ? { candidateId: input.expected.swapped.candidateIds.left, digest: input.expected.swapped.request.left.digest }
124
+ : { candidateId: input.expected.swapped.candidateIds.right, digest: input.expected.swapped.request.right.digest };
125
+ return normalWinner.candidateId === swappedWinner.candidateId && normalWinner.digest === swappedWinner.digest
126
+ ? { kind: 'selected', winnerCandidateId: normalWinner.candidateId, winnerDigest: normalWinner.digest, normal: normal.data, swapped: swapped.data }
127
+ : { kind: 'inconsistent', reason: 'winner-disagrees' };
128
+ }
129
+ function candidatePairMismatch(verdict, expected) {
130
+ if (verdict.attemptId !== expected.request.attemptId)
131
+ return 'attempt-mismatch';
132
+ if (verdict.left.label !== expected.request.left.label || verdict.right.label !== expected.request.right.label)
133
+ return 'label-mismatch';
134
+ if (verdict.left.digest !== expected.request.left.digest || verdict.right.digest !== expected.request.right.digest)
135
+ return 'digest-mismatch';
136
+ if (verdict.provenance.provider !== expected.provenance.provider
137
+ || verdict.provenance.model !== expected.provenance.model
138
+ || verdict.provenance.promptDigest !== expected.provenance.promptDigest
139
+ || verdict.provenance.rubricDigest !== expected.provenance.rubricDigest)
140
+ return 'provenance-mismatch';
141
+ return null;
142
+ }
@@ -2,7 +2,8 @@ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
2
2
  import { join, dirname } from 'node:path';
3
3
  import { parse, stringify } from 'yaml';
4
4
  import { z } from 'zod';
5
- const AgentSchema = z.enum(['claude', 'codex', 'gemini']);
5
+ import { AgentSchema, PermissionProfileSchema } from '../agents/contracts.js';
6
+ import { ProjectQualityDefaultsSchema } from '../quality/types.js';
6
7
  const CodeGraphSchema = z.enum(['graphify', 'serena']);
7
8
  const SmokeFlowSchema = z.object({ name: z.string().min(1), path: z.string().min(1), landmark: z.string().optional() });
8
9
  const SmokeSchema = z.object({ baseUrl: z.string().min(1), flows: z.array(SmokeFlowSchema).min(1) });
@@ -30,7 +31,7 @@ export const YokeConfigSchema = z.object({
30
31
  model: z.string().min(1).optional(),
31
32
  reasoningEffort: z.string().min(1).optional(),
32
33
  bare: z.boolean().optional(),
33
- permissions: z.enum(['safe', 'unsafe', 'read-only']).optional(),
34
+ permissions: PermissionProfileSchema.optional(),
34
35
  }).optional(),
35
36
  routing: z.object({
36
37
  enabled: z.boolean(),
@@ -66,6 +67,7 @@ export const YokeConfigSchema = z.object({
66
67
  perf: z.object({ command: z.string().min(1), retries: z.number().int().nonnegative().optional() }).optional(),
67
68
  codeGraph: CodeGraphSchema.optional(),
68
69
  smoke: SmokeSchema.optional(),
70
+ quality: ProjectQualityDefaultsSchema.optional(),
69
71
  // Opt-in: upgrade yoke at loop START when a newer version is cached (never mid-run).
70
72
  update: z.object({ auto: z.boolean() }).optional(),
71
73
  });
@@ -11,6 +11,8 @@ export const YOKE_IGNORE_LINES = [
11
11
  '.yoke/loop.lock.*.tmp',
12
12
  '.yoke/loop.pause',
13
13
  '.yoke/runner.pid',
14
+ '.yoke/provider-processes/',
15
+ '.yoke/claims/',
14
16
  '.yoke/ambiguity.md',
15
17
  '.yoke/decision-request.yaml',
16
18
  '.yoke/pending-decision.yaml',
@@ -19,6 +21,7 @@ export const YOKE_IGNORE_LINES = [
19
21
  '.yoke/decision-*.yaml.*.tmp',
20
22
  '.yoke/story-durations.json',
21
23
  '.yoke/proof/',
24
+ '.yoke/references/',
22
25
  '.yoke/changes/',
23
26
  ];
24
27
  const HEADER = '# Yoke runtime artifacts (managed by yoke retrofit)';
@@ -1,9 +1,8 @@
1
- import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runReviewAgent, isAgentAvailable, } from '../loop/runner.js';
1
+ import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runCapturedAgent, repositoryFingerprint, isAgentAvailable, } from '../loop/runner.js';
2
+ import { parseProviderResult } from '../agents/telemetry.js';
2
3
  import { resolveIdleMs } from '../loop/run-command.js';
3
- import { existsSync, mkdirSync, rmSync, rmdirSync } from 'node:fs';
4
- import { join } from 'node:path';
5
4
  import { loadConfig } from '../retrofit/config.js';
6
- import { readReviewVerdict, reviewVerdictPath } from './verdict.js';
5
+ import { parseReviewVerdict } from './verdict.js';
7
6
  // Resolve to the first available agent, preferring a *second* model so the review
8
7
  // is genuinely cross-model. claude last => a Claude-only box degrades to self-review.
9
8
  const RESOLUTION_ORDER = ['codex', 'gemini', 'claude'];
@@ -33,51 +32,41 @@ export function runReview(targetDir, opts = {}) {
33
32
  const scope = opts.base
34
33
  ? `the diff ${opts.base}..HEAD`
35
34
  : 'the uncommitted working-tree changes (working tree + staged)';
36
- const verdictPath = reviewVerdictPath(targetDir);
37
- const yokeDir = join(targetDir, '.yoke');
38
- const createdYokeDir = !existsSync(yokeDir);
39
- mkdirSync(yokeDir, { recursive: true });
40
- rmSync(verdictPath, { force: true });
41
- const prompt = buildStandaloneReviewPrompt(scope, opts.focus, verdictPath);
35
+ const prompt = buildStandaloneReviewPrompt(scope, opts.focus, undefined, reviewer);
42
36
  const idleMs = resolveIdleMs(opts.timeoutMinutes, undefined);
43
37
  // Pass the *agent* invocation to the runner so callers (and tests) see the
44
38
  // reviewer command. The default runner adds the watchdog wrapper before exec;
45
39
  // an injected run() gets the raw invocation.
46
- const inv = agentInvocation(reviewer, prompt, targetDir, 'safe');
40
+ const inv = agentInvocation(reviewer, prompt, targetDir, 'read-only');
47
41
  const say = opts.json ? console.error : console.log;
48
42
  say(`Reviewing ${scope} with ${reviewer}...`);
49
- const run = opts.run ?? ((i) => runReviewAgent(buildWatchdogInvocation(i, idleMs)));
43
+ const before = repositoryFingerprint(targetDir);
44
+ const run = opts.run ?? ((i) => runCapturedAgent(reviewer, buildWatchdogInvocation(i, idleMs)));
50
45
  const processResult = run(inv);
51
- let verdict;
52
- try {
53
- verdict = readReviewVerdict(verdictPath);
54
- }
55
- catch (error) {
56
- say(`✗ ${reviewer} produced no valid verdict (${error.message})${processResult.success ? '' : `; process: ${processResult.summary}`}`);
57
- if (createdYokeDir) {
58
- try {
59
- rmdirSync(yokeDir);
60
- }
61
- catch { }
62
- }
46
+ if (repositoryFingerprint(targetDir) !== before) {
47
+ say(`✗ ${reviewer} modified the repository during a read-only review`);
63
48
  return 1;
64
49
  }
65
- if (createdYokeDir) {
66
- try {
67
- rmdirSync(yokeDir);
50
+ try {
51
+ const actualModel = opts.run ? undefined : processResult.tokens?.model;
52
+ if (!opts.run && processResult.success && !actualModel)
53
+ throw new Error('review provider did not report its model');
54
+ const verdict = parseReviewVerdict(parseProviderResult(reviewer, processResult.output), { provider: reviewer, ...(actualModel ? { model: actualModel } : {}) });
55
+ if (opts.json)
56
+ console.log(JSON.stringify({ reviewer, process: processResult, verdict }));
57
+ if (!processResult.success) {
58
+ say(`✗ ${reviewer} process failed (${processResult.summary}); verdict: ${verdict.summary}`);
59
+ return 1;
68
60
  }
69
- catch { }
70
- }
71
- if (opts.json)
72
- console.log(JSON.stringify({ reviewer, process: processResult, verdict }));
73
- if (!processResult.success) {
74
- say(`✗ ${reviewer} process failed (${processResult.summary}); verdict: ${verdict.summary}`);
61
+ if (verdict.approved) {
62
+ say(`✓ ${reviewer} approved: ${verdict.summary}`);
63
+ return 0;
64
+ }
65
+ say(`✗ ${reviewer} rejected: ${verdict.summary}`);
75
66
  return 1;
76
67
  }
77
- if (verdict.approved) {
78
- say(`✓ ${reviewer} approved: ${verdict.summary}`);
79
- return 0;
68
+ catch (error) {
69
+ say(`✗ ${reviewer} produced no valid verdict (${error.message})${processResult.success ? '' : `; process: ${processResult.summary}`}`);
70
+ return 1;
80
71
  }
81
- say(`✗ ${reviewer} rejected: ${verdict.summary}`);
82
- return 1;
83
72
  }
@@ -1,21 +1,49 @@
1
1
  import { existsSync, readFileSync, rmSync } from 'node:fs';
2
2
  import { join, resolve } from 'node:path';
3
3
  import { z } from 'zod';
4
+ import { AgentSchema, PermissionProfileSchema } from '../agents/contracts.js';
4
5
  export const ReviewFindingSchema = z.object({
6
+ id: z.string().min(1).optional(),
5
7
  severity: z.enum(['blocking', 'warning', 'info']),
6
8
  message: z.string().min(1),
7
9
  file: z.string().min(1).optional(),
8
10
  line: z.number().int().positive().optional(),
11
+ actionable: z.boolean().optional(),
12
+ suggestedFix: z.string().min(1).optional(),
13
+ evidence: z.array(z.string().min(1)).optional(),
9
14
  });
10
15
  export const ReviewVerdictSchema = z.object({
16
+ schemaVersion: z.literal(1),
11
17
  approved: z.boolean(),
12
18
  summary: z.string().min(1),
13
19
  findings: z.array(ReviewFindingSchema),
20
+ provenance: z.object({
21
+ provider: AgentSchema,
22
+ model: z.string().min(1),
23
+ role: z.literal('review'),
24
+ promptVersion: z.literal(1),
25
+ permissions: PermissionProfileSchema,
26
+ }),
14
27
  });
28
+ export function parseReviewVerdict(value, expected) {
29
+ const result = ReviewVerdictSchema.safeParse(value);
30
+ if (!result.success)
31
+ throw new Error(`Review verdict is invalid: ${result.error.message}`);
32
+ if (expected && result.data.provenance.provider !== expected.provider)
33
+ throw new Error(`Review verdict provider mismatch: expected ${expected.provider}, received ${result.data.provenance.provider}`);
34
+ if (expected?.model && result.data.provenance.model !== expected.model)
35
+ throw new Error(`Review verdict model mismatch: expected ${expected.model}, received ${result.data.provenance.model}`);
36
+ return result.data;
37
+ }
38
+ export function selectRepairFinding(verdict) {
39
+ if (verdict.approved)
40
+ return null;
41
+ return verdict.findings.find(candidate => candidate.severity === 'blocking' && (candidate.actionable ?? true)) ?? null;
42
+ }
15
43
  export function reviewVerdictPath(targetDir) {
16
44
  return resolve(join(targetDir, '.yoke', 'review-verdict.json'));
17
45
  }
18
- export function readReviewVerdict(path) {
46
+ export function readReviewVerdict(path, expected) {
19
47
  if (!existsSync(path))
20
48
  throw new Error(`Review verdict is missing: ${path}`);
21
49
  try {
@@ -26,20 +54,23 @@ export function readReviewVerdict(path) {
26
54
  catch (error) {
27
55
  throw new Error(`Review verdict is malformed JSON: ${error.message}`);
28
56
  }
29
- const result = ReviewVerdictSchema.safeParse(value);
30
- if (!result.success)
31
- throw new Error(`Review verdict is invalid: ${result.error.message}`);
32
- return result.data;
57
+ return parseReviewVerdict(value, expected);
33
58
  }
34
59
  finally {
35
60
  rmSync(path, { force: true });
36
61
  }
37
62
  }
38
- export function formatReviewContract(path) {
63
+ export function formatReviewStdoutContract(provider) {
64
+ return [
65
+ 'Return exactly one JSON object as your final response. Do not write any file.',
66
+ `{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[],"provenance":{"provider":"${provider}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"read-only"}}`,
67
+ ].join('\n');
68
+ }
69
+ export function formatReviewContract(path, provider) {
39
70
  return [
40
71
  `Write your final verdict to this absolute path: ${path}`,
41
72
  'The file must contain exactly one JSON object with this contract:',
42
- '{"approved":boolean,"summary":"non-empty string","findings":[{"severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1}]}',
73
+ `{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[{"id":"optional id","severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1,"actionable":true,"suggestedFix":"optional repair","evidence":["optional evidence reference"]}],"provenance":{"provider":"${provider ?? 'claude|codex|gemini'}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"safe"}}`,
43
74
  'Set approved=false when any blocking finding exists. Create the file even when the process also exits non-zero.',
44
75
  ].join('\n');
45
76
  }
@@ -0,0 +1,70 @@
1
+ # Migrating to Yoke 1.4
2
+
3
+ Yoke 1.4 adds dependency-aware parallel workers, isolated candidate selection, and a reference-driven quality gauntlet. Existing projects remain serial and skip quality comparison unless you opt in.
4
+
5
+ Upgrade and refresh generated harness files:
6
+
7
+ ```bash
8
+ npm install -g @hecer/yoke@1.4.0
9
+ yoke retrofit .
10
+ ```
11
+
12
+ ## Parallel execution
13
+
14
+ Run dependency-ready stories concurrently with an explicit worker limit:
15
+
16
+ ```bash
17
+ yoke loop run . --isolate --parallel=3
18
+ ```
19
+
20
+ Parallel runs use isolated worktrees, leased claims, and a FIFO integration queue. Every candidate must pass its worker gates and the merged result must pass the project gates again. Adaptive routing is disabled during parallel and multi-candidate runs so each worker has one auditable provider identity.
21
+
22
+ Use `--candidates=N` to produce and mechanically gate multiple implementations before an identity-blind comparison selects the winner. Candidate mode supports up to five candidates.
23
+
24
+ ## Reference-driven quality
25
+
26
+ Quality remains disabled by default. A story must declare a trusted reference, candidate artifacts, and a rubric:
27
+
28
+ ```yaml
29
+ quality:
30
+ reference: { name: approved-home, source: design/home.png, kind: file }
31
+ candidate: { kind: screenshots, paths: [.yoke/proof/STORY-1/home.png] }
32
+ rubric: Match the approved layout, hierarchy, spacing, and states.
33
+ policy: blocking
34
+ ```
35
+
36
+ Project defaults can bound critic and repair work:
37
+
38
+ ```yaml
39
+ quality:
40
+ enabled: false
41
+ policy: blocking
42
+ maxRounds: 3
43
+ maxMinutes: 60
44
+ consistencyChecks: 2
45
+ maxParallelCandidates: 2
46
+ critic: { agent: codex, model: gpt-5.6-sol } # model required for --candidates
47
+ repair: { agent: claude }
48
+ ```
49
+
50
+ Enable it for one run with:
51
+
52
+ ```bash
53
+ yoke loop run . --quality
54
+ ```
55
+
56
+ `consistencyChecks` is fixed at `2`: Yoke runs the swapped-label pair needed to detect identity-sensitive critic output. Advisory mode records both verdicts without blocking; blocking mode permits bounded repairs and reruns all mechanical gates.
57
+
58
+ ## Cleanup behavior
59
+
60
+ `yoke loop cleanup` now retains Yoke-created worktrees by default so failed candidates remain inspectable. Remove them explicitly when no longer needed:
61
+
62
+ ```bash
63
+ yoke loop cleanup . --remove-worktrees
64
+ ```
65
+
66
+ Cleanup still reaps only provider processes recorded for the current project and removes stale loop locks. It never kills providers by machine-wide process name.
67
+
68
+ ## Review verdicts
69
+
70
+ Review verdict files now require `schemaVersion: 1` and provenance containing the provider, provider-reported model, review role, prompt version, and permission profile. Custom reviewer integrations must emit the contract printed in the reviewer prompt. Legacy verdict files without this envelope fail closed.
@@ -1,6 +1,6 @@
1
1
  # Publishing channels — status & playbook
2
2
 
3
- Where Yoke is published, and how each channel gets updated. (Reviewed 2026-07-30.)
3
+ Where Yoke is published, and how each channel gets updated. (Reviewed 2026-08-15.)
4
4
 
5
5
  ## Live
6
6
 
@@ -14,12 +14,26 @@ Where Yoke is published, and how each channel gets updated. (Reviewed 2026-07-30
14
14
 
15
15
  ## GitHub release (required, not just a tag)
16
16
 
17
+ Before the release commit, update every user-facing version and README statistic, then require the
18
+ same checks npm will run:
19
+
20
+ ```bash
21
+ npm run docs:update
22
+ npm run docs:check
23
+ npm run prepublishOnly
24
+ ```
25
+
26
+ `docs:update` synchronizes the README's package version, test count, skill count, and supported
27
+ agents from `package.json`, Vitest discovery, and `canon/manifest.yaml`. The version must also be
28
+ kept in sync in `package-lock.json`, `canon/manifest.yaml`, `.claude-plugin/plugin.json`,
29
+ `.codex-plugin/plugin.json`, and `gemini-extension.json`.
30
+
17
31
  A pushed tag appears under **Tags**, but GitHub only shows an entry under **Releases** after a
18
32
  release object is created. Use this idempotent check after the version commit reaches `main`:
19
33
 
20
34
  ```bash
21
35
  set -euo pipefail
22
- VERSION=1.1.0
36
+ VERSION=1.4.0
23
37
  TARGET=$(git rev-parse HEAD)
24
38
  git fetch --tags origin
25
39