@hecer/yoke 1.21.1 → 1.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +48 -0
  4. package/README.md +8 -1
  5. package/TODOS.md +6 -0
  6. package/bench/analyze-codex-comparison.mjs +90 -17
  7. package/bench/compare-codex.mjs +159 -36
  8. package/bench/result-schema.mjs +132 -0
  9. package/canon/manifest.yaml +1 -1
  10. package/canon/skills/visual-verification/SKILL.md +25 -2
  11. package/canon/tools/codex-rtk-hook.mjs +6 -16
  12. package/dist/agents/pi-telemetry.js +2 -1
  13. package/dist/agents/process-streams.js +12 -64
  14. package/dist/agents/provider-selection.js +12 -0
  15. package/dist/agents/telemetry.js +52 -52
  16. package/dist/change/inbox.js +8 -3
  17. package/dist/check/command.js +69 -17
  18. package/dist/check/delivery.js +121 -0
  19. package/dist/cli.js +91 -3
  20. package/dist/code-intelligence/adapters/mcp.js +1 -0
  21. package/dist/code-intelligence/budgets.js +138 -0
  22. package/dist/code-intelligence/contracts.js +2 -0
  23. package/dist/code-intelligence/coordinator.js +159 -85
  24. package/dist/code-intelligence/evidence.js +87 -34
  25. package/dist/code-intelligence/index.js +1 -0
  26. package/dist/code-intelligence/mcp-client.js +25 -6
  27. package/dist/code-intelligence/mcp-server.js +14 -11
  28. package/dist/code-intelligence/preflight.js +71 -0
  29. package/dist/dashboard/analytics.js +5 -3
  30. package/dist/goals/command.js +183 -53
  31. package/dist/goals/usage.js +87 -0
  32. package/dist/loop/cache-isolation.js +36 -0
  33. package/dist/loop/candidate-cleanup.js +47 -17
  34. package/dist/loop/candidates.js +17 -11
  35. package/dist/loop/dispatcher.js +89 -26
  36. package/dist/loop/failure.js +104 -0
  37. package/dist/loop/gate-snapshot.js +19 -0
  38. package/dist/loop/git.js +1 -1
  39. package/dist/loop/loop.js +124 -70
  40. package/dist/loop/parallel-adapters.js +57 -6
  41. package/dist/loop/parallel-command.js +49 -7
  42. package/dist/loop/proof-retention.js +70 -0
  43. package/dist/loop/recovery.js +23 -5
  44. package/dist/loop/reporter.js +22 -5
  45. package/dist/loop/run-command.js +101 -47
  46. package/dist/loop/runner.js +6 -5
  47. package/dist/loop/worker.js +152 -91
  48. package/dist/observability/history.js +2 -1
  49. package/dist/observability/invocation.js +42 -0
  50. package/dist/observability/local-report.js +120 -0
  51. package/dist/observability/usage.js +19 -0
  52. package/dist/prd/command.js +20 -7
  53. package/dist/prd/decompose.js +5 -2
  54. package/dist/retrofit/config.js +29 -2
  55. package/dist/retrofit/gitignore.js +12 -0
  56. package/dist/retrofit/planners/codex.js +20 -20
  57. package/dist/routing/attempts.js +241 -0
  58. package/dist/routing/capability.js +13 -9
  59. package/dist/routing/optimization.js +73 -0
  60. package/dist/routing/registry.js +7 -1
  61. package/dist/routing/router.js +282 -127
  62. package/dist/setup/command.js +8 -2
  63. package/dist/smoke/command.js +387 -85
  64. package/dist/update/check.js +1 -1
  65. package/docs/BENCHMARK-MANIFEST.md +131 -0
  66. package/docs/CODE-INTELLIGENCE.md +43 -1
  67. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  68. package/docs/DELIVERY-JOURNEYS.md +206 -0
  69. package/docs/ECONOMIC-ROUTING.md +180 -0
  70. package/docs/GOALS.md +61 -4
  71. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  72. package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
  73. package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
  74. package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
  75. package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
  76. package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
  77. package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
  78. package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
  79. package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
  80. package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
  81. package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
  82. package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
  83. package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
  84. package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
  85. package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
  86. package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
  87. package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
  88. package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
  89. package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
  90. package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
  91. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
  92. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
  93. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
  94. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
  95. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
  96. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
  97. package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
  98. package/docs/parallel-execution.md +37 -9
  99. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
  100. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
  101. package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
  102. package/gemini-extension.json +1 -1
  103. package/package.json +1 -1
@@ -1,5 +1,5 @@
1
1
  import { createHash, randomUUID } from 'node:crypto';
2
- import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync } from 'node:fs';
2
+ import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, openSync, readFileSync, readSync, realpathSync, writeFileSync } from 'node:fs';
3
3
  import { homedir } from 'node:os';
4
4
  import { dirname, isAbsolute, join, relative, resolve } from 'node:path';
5
5
  import { parse } from 'yaml';
@@ -15,10 +15,17 @@ import { writeOutputArtifact } from '../output/artifact.js';
15
15
  import { DEFAULT_OUTPUT_POLICY } from '../output/types.js';
16
16
  import { createProviderProcessRecord, filesystemProviderProcessRecordAdapter } from '../agents/process-record.js';
17
17
  import { trackProcessRecordIdentity } from '../agents/process-record-identity.js';
18
+ import { DeliverySchema, startDelivery, finishDelivery } from './delivery.js';
18
19
  const Criterion = z.object({ id: z.string().min(1).max(120), text: z.string().min(1).max(8000), commands: z.array(z.string().min(1).max(8000)).max(30) }).strict();
19
- const Acceptance = z.object({ version: z.literal(1), criteria: z.array(Criterion).max(200), protected: z.array(z.string().min(1)).max(500).default([]) }).strict().superRefine((value, ctx) => {
20
+ const Acceptance = z.object({ version: z.literal(1), criteria: z.array(Criterion).max(200), protected: z.array(z.string().min(1)).max(500).default([]), delivery: DeliverySchema.optional() }).strict().superRefine((value, ctx) => {
20
21
  if (new Set(value.criteria.map(c => c.id)).size !== value.criteria.length)
21
22
  ctx.addIssue({ code: 'custom', message: 'Duplicate acceptance criterion id' });
23
+ const ids = new Set(value.criteria.map(criterion => criterion.id));
24
+ for (const item of [...(value.delivery?.artifacts ?? []), ...(value.delivery?.journeys ?? [])]) {
25
+ for (const id of item.criteria)
26
+ if (!ids.has(id))
27
+ ctx.addIssue({ code: 'custom', path: ['delivery'], message: `Delivery references unknown criterion: ${id}` });
28
+ }
22
29
  });
23
30
  export async function checkProjectAsync(directory, options = {}) {
24
31
  const controller = new AbortController();
@@ -162,8 +169,36 @@ function verifyCommandAsync(command, root, signal) {
162
169
  });
163
170
  }
164
171
  export function loadAcceptance(root) {
172
+ return loadAcceptanceSnapshot(root).manifest;
173
+ }
174
+ /** Hash the exact bounded bytes that were parsed; never reread an untrusted path for the digest. */
175
+ function loadAcceptanceSnapshot(root) {
165
176
  const file = statePath(root, 'acceptance.yaml');
166
- return existsSync(file) ? Acceptance.parse(parse(readFileSync(file, 'utf8'))) : null;
177
+ if (!existsSync(file))
178
+ return { manifest: null, digest: null };
179
+ const fd = openSync(file, constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0) | (constants.O_NONBLOCK ?? 0));
180
+ try {
181
+ const before = fstatSync(fd);
182
+ if (!before.isFile() || before.size > 64 * 1024 * 1024)
183
+ throw Error('Acceptance must be a regular file of at most 64 MiB');
184
+ const bytes = Buffer.alloc(before.size + 1);
185
+ let length = 0;
186
+ while (length < bytes.length) {
187
+ const count = readSync(fd, bytes, length, bytes.length - length, null);
188
+ if (!count)
189
+ break;
190
+ length += count;
191
+ }
192
+ const after = fstatSync(fd), named = lstatSync(statePath(root, 'acceptance.yaml'));
193
+ if (length !== before.size || after.size !== before.size || after.mtimeMs !== before.mtimeMs || after.ctimeMs !== before.ctimeMs ||
194
+ named.ino !== before.ino || named.dev !== before.dev || named.size !== before.size || named.mtimeMs !== before.mtimeMs || named.ctimeMs !== before.ctimeMs)
195
+ throw Error('Acceptance changed while loading');
196
+ const content = bytes.subarray(0, length);
197
+ return { manifest: Acceptance.parse(parse(content.toString('utf8'))), digest: createHash('sha256').update(content).digest('hex') };
198
+ }
199
+ finally {
200
+ closeSync(fd);
201
+ }
167
202
  }
168
203
  function protectedPath(root, path) {
169
204
  if (isAbsolute(path))
@@ -209,17 +244,41 @@ export function acceptanceProtectionProblem(root, baselineRoot = root) {
209
244
  return `Protected acceptance cannot be verified: ${error.message}`;
210
245
  }
211
246
  }
247
+ function finalizeDelivery(root, manifest, start, fingerprint, criteria, options = {}) {
248
+ let changed = false;
249
+ const delivery = finishDelivery(root, manifest, start, fingerprint, criteria, { ...options, checkIntegrity: () => {
250
+ let problem = null;
251
+ try {
252
+ changed = workspaceFingerprint(root) !== fingerprint;
253
+ }
254
+ catch (error) {
255
+ changed = true;
256
+ problem = `Checked source cannot be verified: ${error.message}`;
257
+ }
258
+ problem = problem ?? acceptanceProtectionProblem(root) ?? (changed ? 'Source changed during verification; run check again on a stable tree' : null);
259
+ if (problem)
260
+ criteria.push({ id: 'source-integrity', text: 'Checked source remained stable', commands: [], status: 'failed', summary: problem });
261
+ if (options.signal?.aborted || (options.deadline !== undefined && Date.now() >= options.deadline)) {
262
+ const cancelled = 'Verification cancelled or deadline reached';
263
+ criteria.push({ id: 'verification-cancelled', text: 'Verification completed within its allowed run', commands: [], status: 'failed', summary: cancelled });
264
+ problem = problem ?? cancelled;
265
+ }
266
+ return problem;
267
+ } });
268
+ return { delivery, changed };
269
+ }
212
270
  export function checkProject(directory, options = {}) {
213
271
  const root = realpathSync(directory);
214
272
  const started = Date.now();
215
273
  const before = workspaceFingerprint(root);
216
274
  const problem = acceptanceProtectionProblem(root);
275
+ const { manifest, digest } = problem ? { manifest: null, digest: null } : loadAcceptanceSnapshot(root);
276
+ const deliveryStart = startDelivery(root, manifest, digest);
217
277
  const criteria = [];
218
278
  const execute = options.execute ?? ((command, cwd) => commandVerifier(command, { phase: 'verify' })(cwd));
219
279
  if (problem)
220
280
  criteria.push({ id: 'protected-acceptance', text: 'Acceptance infrastructure unchanged', commands: [], status: 'failed', summary: problem });
221
281
  else {
222
- const manifest = loadAcceptance(root);
223
282
  for (const criterion of manifest?.criteria ?? []) {
224
283
  const results = criterion.commands.map(command => {
225
284
  try {
@@ -247,14 +306,11 @@ export function checkProject(directory, options = {}) {
247
306
  if (criteria.length === 0)
248
307
  criteria.push({ id: 'acceptance', text: 'Project acceptance', commands: [], status: 'unverified', summary: 'No acceptance manifest or project verification command found' });
249
308
  }
250
- const changed = workspaceFingerprint(root) !== before;
251
- const afterProblem = acceptanceProtectionProblem(root);
252
- if (changed || afterProblem)
253
- criteria.push({ id: 'source-integrity', text: 'Checked source remained stable', commands: [], status: 'failed', summary: afterProblem ?? 'Source changed during verification; run check again on a stable tree' });
309
+ const { changed, delivery } = finalizeDelivery(root, manifest, deliveryStart, before, criteria);
254
310
  const status = criteria.some(c => c.status === 'failed') ? 'failed' : criteria.some(c => c.status === 'unverified') ? 'unverified' : 'passed';
255
311
  const id = randomUUID();
256
312
  const evidencePath = statePath(root, 'checks', `${id}.json`);
257
- const report = { version: 1, id, generatedAt: new Date().toISOString(), fingerprint: before, status, summary: changed ? 'Source changed during verification' : `${criteria.filter(c => c.status === 'passed').length}/${criteria.length} checks passed; ${status}`, criteria, durationMs: Date.now() - started, evidencePath };
313
+ const report = { version: 1, id, generatedAt: new Date().toISOString(), fingerprint: before, status, summary: changed ? 'Source changed during verification' : `${criteria.filter(c => c.status === 'passed').length}/${criteria.length} checks passed; ${status}`, criteria, delivery, durationMs: Date.now() - started, evidencePath };
258
314
  mkdirSync(dirname(evidencePath), { recursive: true });
259
315
  writeFileSync(evidencePath, JSON.stringify(report, null, 2) + '\n', { flag: 'wx', mode: 0o600 });
260
316
  return report;
@@ -265,6 +321,8 @@ async function checkProjectAsyncAdmitted(directory, options) {
265
321
  const started = Date.now();
266
322
  const before = workspaceFingerprint(root);
267
323
  const problem = acceptanceProtectionProblem(root);
324
+ const { manifest, digest } = problem ? { manifest: null, digest: null } : loadAcceptanceSnapshot(root);
325
+ const deliveryStart = startDelivery(root, manifest, digest, options);
268
326
  const criteria = [];
269
327
  let cleanupUnconfirmed = false;
270
328
  const rawExecute = options.execute ?? ((command, cwd) => verifyCommandAsync(command, cwd, options.signal));
@@ -281,7 +339,6 @@ async function checkProjectAsyncAdmitted(directory, options) {
281
339
  if (problem)
282
340
  criteria.push({ id: 'protected-acceptance', text: 'Acceptance infrastructure unchanged', commands: [], status: 'failed', summary: problem });
283
341
  else {
284
- const manifest = loadAcceptance(root);
285
342
  for (const criterion of manifest?.criteria ?? []) {
286
343
  const results = [];
287
344
  for (const command of criterion.commands) {
@@ -316,16 +373,11 @@ async function checkProjectAsyncAdmitted(directory, options) {
316
373
  if (criteria.length === 0)
317
374
  criteria.push({ id: 'acceptance', text: 'Project acceptance', commands: [], status: 'unverified', summary: 'No acceptance manifest or project verification command found' });
318
375
  }
319
- const changed = workspaceFingerprint(root) !== before;
320
- const afterProblem = acceptanceProtectionProblem(root);
321
- if (changed || afterProblem)
322
- criteria.push({ id: 'source-integrity', text: 'Checked source remained stable', commands: [], status: 'failed', summary: afterProblem ?? 'Source changed during verification; run check again on a stable tree' });
323
- if (options.signal?.aborted)
324
- criteria.push({ id: 'verification-cancelled', text: 'Verification completed within its allowed run', commands: [], status: 'failed', summary: 'Verification cancelled or deadline reached' });
376
+ const { changed, delivery } = finalizeDelivery(root, manifest, deliveryStart, before, criteria, options);
325
377
  const status = criteria.some(c => c.status === 'failed') ? 'failed' : criteria.some(c => c.status === 'unverified') ? 'unverified' : 'passed';
326
378
  const id = randomUUID();
327
379
  const evidencePath = statePath(root, 'checks', `${id}.json`);
328
- const report = { ...(cleanupUnconfirmed ? { cleanupUnconfirmed: true } : {}), version: 1, id, generatedAt: new Date().toISOString(), fingerprint: before, status, summary: changed ? 'Source changed during verification' : `${criteria.filter(c => c.status === 'passed').length}/${criteria.length} checks passed; ${status}`, criteria, durationMs: Date.now() - started, evidencePath };
380
+ const report = { ...(cleanupUnconfirmed ? { cleanupUnconfirmed: true } : {}), version: 1, id, generatedAt: new Date().toISOString(), fingerprint: before, status, summary: changed ? 'Source changed during verification' : `${criteria.filter(c => c.status === 'passed').length}/${criteria.length} checks passed; ${status}`, criteria, delivery, durationMs: Date.now() - started, evidencePath };
329
381
  mkdirSync(dirname(evidencePath), { recursive: true });
330
382
  writeFileSync(evidencePath, JSON.stringify(report, null, 2) + '\n', { flag: 'wx', mode: 0o600 });
331
383
  return report;
@@ -0,0 +1,121 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { closeSync, constants, fstatSync, lstatSync, openSync, readSync, realpathSync } from 'node:fs';
3
+ import { isAbsolute, join, relative, resolve, sep } from 'node:path';
4
+ import { z } from 'zod';
5
+ const References = z.array(z.string().min(1).max(120)).min(1).max(200);
6
+ export const DeliverySchema = z.object({
7
+ version: z.literal(1),
8
+ artifacts: z.array(z.object({ path: z.string().min(1).max(1000), criteria: References }).strict()).max(32).default([]),
9
+ journeys: z.array(z.object({ id: z.string().min(1).max(120), text: z.string().min(1).max(8000), criteria: References }).strict()).max(100).default([]),
10
+ environment: z.object({ name: z.string().min(1).max(200), url: z.string().url().max(2000).optional() }).strict().optional(),
11
+ }).strict().superRefine((value, context) => {
12
+ if (new Set(value.artifacts.map(item => item.path)).size !== value.artifacts.length)
13
+ context.addIssue({ code: 'custom', message: 'Duplicate delivery artifact path' });
14
+ if (new Set(value.journeys.map(item => item.id)).size !== value.journeys.length)
15
+ context.addIssue({ code: 'custom', message: 'Duplicate delivery journey id' });
16
+ });
17
+ export const DELIVERY_HASH_LIMITS = Object.freeze({ fileBytes: 512 * 1024 * 1024, totalBytes: 1024 * 1024 * 1024, timeoutMs: 10_000 });
18
+ function snapshotBudget(options) {
19
+ return { signal: options.signal, remaining: DELIVERY_HASH_LIMITS.totalBytes, deadline: Math.min(options.deadline ?? Infinity, Date.now() + DELIVERY_HASH_LIMITS.timeoutMs) };
20
+ }
21
+ function checkBudget(budget) {
22
+ if (budget.signal?.aborted || Date.now() >= budget.deadline)
23
+ throw Error('Artifact hashing cancelled or deadline reached');
24
+ }
25
+ function artifactPath(root, path) {
26
+ if (isAbsolute(path))
27
+ throw Error('Artifact path must be relative');
28
+ const full = resolve(root, path), rel = relative(root, full);
29
+ if (!rel || rel === '..' || rel.startsWith(`..${sep}`) || isAbsolute(rel))
30
+ throw Error('Artifact path escapes project');
31
+ let current = root;
32
+ for (const component of rel.split(sep)) {
33
+ current = join(current, component);
34
+ if (lstatSync(current).isSymbolicLink())
35
+ throw Error('Artifact paths must not contain symbolic links');
36
+ }
37
+ if (realpathSync(full) !== full)
38
+ throw Error('Artifact path is not canonical');
39
+ return full;
40
+ }
41
+ /** Bounded bytes and constant memory; the deadline is checked between synchronous reads. */
42
+ function snapshotArtifact(root, path, budget) {
43
+ let fd;
44
+ try {
45
+ checkBudget(budget);
46
+ const full = artifactPath(root, path);
47
+ const named = lstatSync(full);
48
+ if (!named.isFile())
49
+ throw Error('Artifact must be a regular file');
50
+ if (named.size > DELIVERY_HASH_LIMITS.fileBytes)
51
+ throw Error(`Artifact exceeds the ${DELIVERY_HASH_LIMITS.fileBytes} byte file limit`);
52
+ if (named.size > budget.remaining)
53
+ throw Error(`Artifact exceeds the remaining aggregate hash budget (${budget.remaining} bytes)`);
54
+ fd = openSync(full, constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0) | (constants.O_NONBLOCK ?? 0));
55
+ const before = fstatSync(fd);
56
+ if (!before.isFile() || before.ino !== named.ino || before.dev !== named.dev || before.size !== named.size)
57
+ throw Error('Artifact changed before hashing');
58
+ const hash = createHash('sha256'), buffer = Buffer.alloc(256 * 1024);
59
+ let bytes = 0;
60
+ while (bytes < before.size) {
61
+ checkBudget(budget);
62
+ const count = readSync(fd, buffer, 0, Math.min(buffer.length, before.size - bytes), null);
63
+ if (!count)
64
+ break;
65
+ hash.update(buffer.subarray(0, count));
66
+ bytes += count;
67
+ budget.remaining -= count;
68
+ }
69
+ checkBudget(budget);
70
+ const after = fstatSync(fd);
71
+ const finalNamed = lstatSync(artifactPath(root, path));
72
+ if (bytes !== before.size || after.size !== before.size || after.mtimeMs !== before.mtimeMs || after.ctimeMs !== before.ctimeMs ||
73
+ !finalNamed.isFile() || finalNamed.ino !== before.ino || finalNamed.dev !== before.dev || finalNamed.size !== before.size || finalNamed.mtimeMs !== before.mtimeMs || finalNamed.ctimeMs !== before.ctimeMs)
74
+ throw Error('Artifact changed while hashing');
75
+ return { sha256: hash.digest('hex'), bytes };
76
+ }
77
+ catch (error) {
78
+ return { problem: `Artifact cannot be bound: ${error.message}` };
79
+ }
80
+ finally {
81
+ if (fd !== undefined) {
82
+ try {
83
+ closeSync(fd);
84
+ }
85
+ catch { /* Preserve the snapshot failure. */ }
86
+ }
87
+ }
88
+ }
89
+ export function startDelivery(root, manifest, acceptanceDigest, options = {}) {
90
+ const budget = snapshotBudget(options);
91
+ return { acceptanceDigest, artifacts: (manifest?.delivery?.artifacts ?? []).map(item => snapshotArtifact(root, item.path, budget)) };
92
+ }
93
+ export function finishDelivery(root, manifest, start, fingerprint, criteria, options = {}) {
94
+ const budget = snapshotBudget(options);
95
+ const snapshots = (manifest?.delivery?.artifacts ?? []).map((artifact, index) => {
96
+ const before = start.artifacts[index], after = snapshotArtifact(root, artifact.path, budget);
97
+ const stable = !!before?.sha256 && before.sha256 === after.sha256;
98
+ const problem = before?.problem ?? after.problem ?? (!stable ? 'Artifact changed during verification; rebuild and check the stable artifact again' : undefined);
99
+ if (problem)
100
+ criteria.push({ id: `delivery-artifact-${index + 1}`, text: `Artifact ${artifact.path} remained stable`, commands: [], status: 'failed', summary: problem });
101
+ return { artifact, before, after, stable, problem };
102
+ });
103
+ // Recheck source and cancellation after the final artifact I/O, before assigning proof status.
104
+ const bindingProblem = options.checkIntegrity?.() ?? snapshots.find(item => item.problem)?.problem;
105
+ const statusFor = (ids) => {
106
+ const statuses = ids.flatMap(id => { const matches = criteria.filter(criterion => criterion.id === id); return matches.length ? matches.map(item => item.status) : ['unverified']; });
107
+ return statuses.some(status => status === 'failed') ? 'failed' : !bindingProblem && statuses.every(status => status === 'passed') ? 'passed' : 'unverified';
108
+ };
109
+ const artifacts = snapshots.map(({ artifact, before, after, stable, problem }) => {
110
+ return { ...artifact, binding: 'declared-criteria', status: problem ? 'failed' : statusFor(artifact.criteria), stable, sha256: before?.sha256, bytes: before?.bytes, afterSha256: after.sha256, summary: problem ?? bindingProblem ?? 'Artifact hashes matched before and after checking its declared criteria; the project commands define how this artifact is exercised' };
111
+ });
112
+ const requirements = manifest?.criteria.map(item => item.id) ?? [];
113
+ return {
114
+ version: 1, sourceFingerprint: fingerprint, acceptanceDigest: start.acceptanceDigest, ...(bindingProblem ? { bindingProblem } : {}),
115
+ environment: { platform: process.platform, architecture: process.arch, nodeVersion: process.version, declared: manifest?.delivery?.environment },
116
+ artifacts, journeys: (manifest?.delivery?.journeys ?? []).map(journey => ({ ...journey, status: statusFor(journey.criteria) })),
117
+ passedRequirementIds: requirements.filter(id => statusFor([id]) === 'passed'),
118
+ failedRequirementIds: requirements.filter(id => statusFor([id]) === 'failed'),
119
+ unverifiedRequirementIds: requirements.filter(id => statusFor([id]) === 'unverified'),
120
+ };
121
+ }
package/dist/cli.js CHANGED
@@ -5,7 +5,7 @@ import { realpathSync } from 'node:fs';
5
5
  import { checkProject, checkExitCode, protectAcceptance } from './check/command.js';
6
6
  import { registerProject, listProjects, unregisterProject } from './dashboard/registry.js';
7
7
  import { startDashboard } from './dashboard/server.js';
8
- import { createProjectGoal, readProjectGoal, runProjectGoal, pauseProjectGoal, goalHandoff, budgetProjectGoal, bindProjectGoal } from './goals/command.js';
8
+ import { createProjectGoal, readProjectGoal, runProjectGoal, pauseProjectGoal, goalHandoff, budgetProjectGoal, bindProjectGoal, assessProjectGoal } from './goals/command.js';
9
9
  import { validateCanon } from './canon/validate.js';
10
10
  import { runRetrofit } from './retrofit/command.js';
11
11
  import { setLoopEnabled, loopStatus, runLoopCommand } from './loop/run-command.js';
@@ -26,6 +26,8 @@ import { printAudit, runAudit } from './audit/command.js';
26
26
  import { runSetup } from './setup/command.js';
27
27
  import { runCodeIntelligenceServer } from './code-intelligence/mcp-server.js';
28
28
  import { pendingChanges, queueChange } from './change/inbox.js';
29
+ import { localUsageReport } from './observability/local-report.js';
30
+ import { runToolPreflight } from './code-intelligence/preflight.js';
29
31
  import { answerPendingDecision, answeredDecisionResumeIsValid, clearDecisionResume, decisionProcessingExists, decisionResumeMatchesCurrent, finalizeCommittedDecisionResume, formatPendingDecision, readDecisionResume, readPendingDecision, writeDecisionResume, } from './loop/decision.js';
30
32
  export { runRetrofit } from './retrofit/command.js';
31
33
  export function runValidate(canonDir) {
@@ -149,7 +151,83 @@ function parseExploreLimitFlag(args, explore) {
149
151
  }
150
152
  export function main(argv) {
151
153
  const [cmd, ...rest] = argv;
154
+ // Help must exit before any command can write configuration or start a process.
155
+ if (argv.includes('--help') || argv.includes('-h'))
156
+ return main([]);
157
+ const mutationFlags = {
158
+ setup: ['--yes', '--host=', '--agent=', '--runner=', '--code-graph=', '--code-intelligence=', '--decision-policy=', '--loop', '--no-loop', '--routing', '--no-routing', '--routing-strategy=', '--model-provider=', '--runner-model=', '--runner-reasoning=', '--model=', '--reasoning=', '--clean-worktrees', '--configure-models', '--routing-preset'],
159
+ retrofit: ['--loop', '--agent=', '--code-graph=', '--code-intelligence=', '--clean-worktrees', '--runner-model=', '--runner-reasoning=', '--model=', '--reasoning='],
160
+ context: [],
161
+ projects: [],
162
+ worktrees: ['--all', '--force'],
163
+ dashboard: ['--port=', '--no-register'],
164
+ goal: ['--objective=', '--attempts=', '--minutes=', '--wall-minutes=', '--tokens=', '--criteria=', '--clear-token-budget', '--runner=', '--model=', '--effort=', '--bare', '--native-goal', '--no-native-goal'],
165
+ check: ['--json', '--protect', '--refresh', '--requirement='],
166
+ change: ['--idea='],
167
+ loop: ['--compact', '--remove-worktrees', '--discard-stale-recovery', '--discard', '--explore', '--explore-interval=', '--explore-limit=', '--quality', '--no-quality', '--quality-rounds=', '--quality-minutes=', '--quality-policy=', '--quality-unbounded', '--candidates=', '--choice=', '--rationale=', '--no-resume', '--max=', '--runner=', '--reviewer=', '--review', '--allow-self-review', '--routing', '--no-routing', '--isolate', '--no-isolate', '--unsafe', '--timeout=', '--decision-policy=', '--parallel=', '--json', '--on-ambiguity=', '--resume-worktree'],
168
+ new: ['--idea=', '--agent=', '--runner=', '--loop'],
169
+ prd: ['--idea=', '--runner=', '--story=', '--apply', '--reassess', '--force', '--timeout='],
170
+ review: ['--reviewer=', '--base=', '--focus=', '--allow-self-review', '--json', '--timeout='],
171
+ audit: ['--json'],
172
+ 'flow-smoke': ['--url=', '--label='],
173
+ 'design-scan': ['--report', '--max='],
174
+ usage: ['--json', '--from=', '--to=', '--run='],
175
+ 'tools-preflight': ['--json'],
176
+ };
177
+ const accepted = mutationFlags[cmd];
178
+ if (accepted) {
179
+ const unknown = rest.find(value => value.startsWith('-') && !accepted.some(flag => flag.endsWith('=') ? value.startsWith(flag) && value.length > flag.length : value === flag));
180
+ if (unknown) {
181
+ console.error(`Unknown or incomplete ${cmd} flag: ${unknown}`);
182
+ return 1;
183
+ }
184
+ for (const toggle of ['loop', 'routing', 'isolate', 'native-goal', 'quality']) {
185
+ if (rest.includes(`--${toggle}`) && rest.includes(`--no-${toggle}`)) {
186
+ console.error(`Cannot use --${toggle} with --no-${toggle}.`);
187
+ return 1;
188
+ }
189
+ }
190
+ }
152
191
  switch (cmd) {
192
+ case 'tools-preflight': {
193
+ return runToolPreflight(rest.find(arg => !arg.startsWith('-')) ?? '.').then(report => {
194
+ if (rest.includes('--json'))
195
+ console.log(JSON.stringify(report));
196
+ else {
197
+ console.log(`RTK: ${report.rtk.status}; native rewrite=${report.rtk.nativeRewrite}; nested code mode=${report.rtk.nestedCodeMode}`);
198
+ console.log(`Code intelligence: ${report.codeIntelligence.status}; mode=${report.codeIntelligence.mode}; workspace=${report.codeIntelligence.workspaceId}`);
199
+ console.log(` ${report.rtk.fallback}\n ${report.codeIntelligence.fallback}; index freshness=${report.codeIntelligence.indexFreshness}`);
200
+ }
201
+ return 0;
202
+ }).catch(error => { console.error(`Tool preflight unavailable: ${error.message}`); return 2; });
203
+ }
204
+ case 'usage': {
205
+ const value = (name) => rest.find(arg => arg.startsWith(`--${name}=`))?.slice(name.length + 3);
206
+ const to = value('to') === undefined ? Date.now() : Date.parse(value('to'));
207
+ const from = value('from') === undefined ? to - 30 * 86400000 : Date.parse(value('from'));
208
+ if (!Number.isFinite(from) || !Number.isFinite(to) || from >= to || to - from > 366 * 86400000) {
209
+ console.error('Invalid usage time range: choose --from/--to dates spanning at most 366 days.');
210
+ return 2;
211
+ }
212
+ try {
213
+ const report = localUsageReport(rest.find(arg => !arg.startsWith('-')) ?? '.', { from, to, runId: value('run') });
214
+ if (rest.includes('--json'))
215
+ console.log(JSON.stringify(report));
216
+ else {
217
+ console.log(`Usage ${report.from} to ${report.to}: input=${report.total.inputTokens ?? 'unknown'} cache=${report.total.cachedInputTokens ?? 'unknown'} output=${report.total.outputTokens ?? 'unknown'} cost=${report.total.totalCostUsd ?? 'unknown'} (${report.total.coverage}; cost ${report.total.costCoverage})`);
218
+ console.log(`Guardian=${report.hostCoverage.guardian} approval=${report.hostCoverage.approval}; phase sum=${report.time.phaseDurationSumMs ?? 'unknown'}ms union=${report.time.phaseDurationUnionMs ?? 'unknown'}ms`);
219
+ for (const limit of report.limitations)
220
+ console.log(` ${limit}`);
221
+ for (const error of report.errors)
222
+ console.error(`Usage history: ${error}`);
223
+ }
224
+ return report.errors.length ? 2 : 0;
225
+ }
226
+ catch (error) {
227
+ console.error(`Usage report unavailable: ${error.message}`);
228
+ return 2;
229
+ }
230
+ }
153
231
  case 'code-intelligence-server':
154
232
  return runCodeIntelligenceServer(rest).then(() => 0).catch(error => { console.error(`Code intelligence server: ${error.message}`); return 2; });
155
233
  case 'setup': {
@@ -339,6 +417,15 @@ export function main(argv) {
339
417
  console.log('Pause requested at next safe boundary');
340
418
  return 0;
341
419
  }
420
+ if (sub === 'assess') {
421
+ const provider = value('runner');
422
+ if (provider && !SUPPORTED_AGENTS.includes(provider))
423
+ throw new Error('Unknown runner');
424
+ return assessProjectGoal(targetDir, { provider: provider, selection: { model: value('model'), reasoningEffort: value('effort'), bare: rest.includes('--bare') ? true : undefined } }).then(result => {
425
+ console.log(JSON.stringify(result));
426
+ return result.assessed ? 0 : 1;
427
+ }).catch(error => { console.error(`Goal assessment: ${error.message}`); return 2; });
428
+ }
342
429
  if (sub === 'run' || sub === 'resume') {
343
430
  const provider = value('runner');
344
431
  if (provider && !SUPPORTED_AGENTS.includes(provider))
@@ -348,7 +435,7 @@ export function main(argv) {
348
435
  return goal.status === 'complete' ? 0 : 1;
349
436
  }).catch(error => { console.error(`Goal: ${error.message}`); return 2; });
350
437
  }
351
- throw new Error('Use goal set|bind|status|run|resume|pause|handoff|budget');
438
+ throw new Error('Use goal set|bind|assess|status|run|resume|pause|handoff|budget');
352
439
  }
353
440
  catch (error) {
354
441
  console.error(`Goal: ${error.message}`);
@@ -832,7 +919,8 @@ export function main(argv) {
832
919
  case 'upgrade':
833
920
  return runUpgrade();
834
921
  default:
835
- console.log('Project workflows: yoke check [dir] [--json|--protect] | goal set|run|resume|pause|status|handoff|budget [dir] | projects add|list|remove | dashboard [dir] [--port=N]');
922
+ console.log('Local diagnostics: yoke tools-preflight [dir] [--json] | usage [dir] [--from=ISO] [--to=ISO] [--run=id] [--json]');
923
+ console.log('Project workflows: yoke check [dir] [--json|--protect] | goal set|bind|assess|run|resume|pause|status|handoff|budget [dir] | projects add|list|remove | dashboard [dir] [--port=N]');
836
924
  console.log(`usage: yoke <setup [dir] | new <dir> [--idea="..."] | validate [canonDir] | retrofit [targetDir] [--agent=${AGENT_LIST}|all] [--code-graph=graphify|serena] [--code-intelligence=off|shadow|active] [--clean-worktrees] [--runner-model=<model>] [--runner-reasoning=<effort>] [--model=<agent>:<model>] [--reasoning=<agent>:<effort>] [--loop] | worktrees <list|prune> [dir] [--all] [--force] | change <add|status> [dir] | code-intelligence-server --workspace=<dir> --mode=<mode> | prd <draft|check|assess|decompose> [dir] | loop <on|off|status|decision|answer|resume|run|cleanup> | context <init|status> | review [dir] | design-scan [dir] | flow-smoke [dir] | upgrade>`);
837
925
  return cmd ? 1 : 0;
838
926
  }
@@ -54,5 +54,6 @@ export class McpBackendAdapter {
54
54
  const tool = this.aliases[request.tool] ?? request.tool;
55
55
  return valueFromResult(await this.client.call(tool, request.arguments, timeoutMs));
56
56
  }
57
+ probe(timeoutMs) { return this.client.listTools(timeoutMs); }
57
58
  close() { return this.client.close(); }
58
59
  }
@@ -0,0 +1,138 @@
1
+ import { shorten } from './evidence.js';
2
+ const TRUNCATED = 'BUDGET_EXCEEDED: result truncated; coverage is partial.';
3
+ /** The provider/tokenizer is unknown to this facade. Count one budget unit per
4
+ * UTF-8 byte conservatively, including the response's own metrics. This is not
5
+ * provider-measured token usage. JSON-RPC transport framing is outside this
6
+ * structuredContent response and is not included. */
7
+ export function measureResponse(response) {
8
+ response.metrics.token_count_kind = 'estimated';
9
+ response.metrics.token_count_method = 'utf8_bytes_conservative';
10
+ for (let i = 0; i < 10; i++) {
11
+ const bytes = Buffer.byteLength(JSON.stringify(response), 'utf8');
12
+ if (response.metrics.returned_bytes === bytes && response.metrics.result_tokens === bytes)
13
+ return response;
14
+ response.metrics.returned_bytes = bytes;
15
+ response.metrics.result_tokens = bytes;
16
+ }
17
+ return response;
18
+ }
19
+ export function responseByteLimit(limits) {
20
+ return Math.max(0, Math.floor(Math.min(limits.tokenBudget, limits.maxBytes)));
21
+ }
22
+ /** This fixed-size failure is the only exception when a caller's budget cannot
23
+ * even represent the required error envelope. It never echoes backend content. */
24
+ export function budgetError(response) {
25
+ return measureResponse({
26
+ schema_version: response.schema_version, request_id: shorten(response.request_id, 64), workspace_id: shorten(response.workspace_id, 128), snapshot_id: shorten(response.snapshot_id, 128),
27
+ status: 'error', data: null,
28
+ coverage: { backends_requested: [], backends_used: [], backends_missing: [], structural: 'unavailable', semantic: 'unavailable', documents: 'unavailable' },
29
+ provenance: [], warnings: [], metrics: { latency_ms: response.metrics.latency_ms, result_tokens: 0, token_count_kind: 'estimated', returned_bytes: 0 }, artifact_uri: null,
30
+ error: { code: 'BUDGET_EXCEEDED', message: 'Budget cannot represent the mandatory response envelope.' },
31
+ });
32
+ }
33
+ function markPartial(response) {
34
+ if (response.status === 'success')
35
+ response.status = 'partial';
36
+ for (const key of ['structural', 'semantic', 'documents'])
37
+ if (response.coverage[key] !== 'unavailable')
38
+ response.coverage[key] = 'partial';
39
+ }
40
+ function usedEvidence(value, ids = new Set()) {
41
+ if (Array.isArray(value))
42
+ for (const entry of value)
43
+ usedEvidence(entry, ids);
44
+ else if (value && typeof value === 'object')
45
+ for (const [key, entry] of Object.entries(value)) {
46
+ if (key === 'evidence_ids' && Array.isArray(entry)) {
47
+ for (const id of entry)
48
+ if (typeof id === 'string')
49
+ ids.add(id);
50
+ }
51
+ else
52
+ usedEvidence(entry, ids);
53
+ }
54
+ return ids;
55
+ }
56
+ function pruneProvenance(response) {
57
+ const ids = usedEvidence(response.data);
58
+ response.provenance = response.provenance.filter(item => item.evidence_id && ids.has(item.evidence_id));
59
+ }
60
+ function shortenLargestText(value) {
61
+ const fields = [];
62
+ function visit(node) {
63
+ if (Array.isArray(node))
64
+ for (const item of node)
65
+ visit(item);
66
+ else if (node && typeof node === 'object')
67
+ for (const [key, entry] of Object.entries(node)) {
68
+ if (['excerpt', 'signature', 'message', 'label'].includes(key) && typeof entry === 'string' && entry.length > 64)
69
+ fields.push({ object: node, key, value: entry });
70
+ else if (entry && typeof entry === 'object')
71
+ visit(entry);
72
+ }
73
+ }
74
+ visit(value);
75
+ const field = fields.sort((a, b) => Buffer.byteLength(b.value) - Buffer.byteLength(a.value))[0];
76
+ if (!field)
77
+ return false;
78
+ field.object[field.key] = shorten(field.value, Math.max(63, Math.floor(field.value.length / 2))) + '…';
79
+ return true;
80
+ }
81
+ function removeResultEntry(value) {
82
+ if (!value || typeof value !== 'object' || Array.isArray(value))
83
+ return false;
84
+ const data = value;
85
+ // All facade result collections remain arrays; identifiers and operation
86
+ // receipts stay intact. Never replace typed data with a prose excerpt.
87
+ const keys = ['items', 'symbols', 'references', 'diagnostics', 'nodes', 'edges', 'semantic_findings', 'structural_candidates', 'documents', 'suggested_test_paths', 'unresolved', 'changed_paths'];
88
+ const candidates = keys.filter(key => Array.isArray(data[key]) && data[key].length > 0);
89
+ candidates.sort((a, b) => Buffer.byteLength(JSON.stringify(data[b].at(-1))) - Buffer.byteLength(JSON.stringify(data[a].at(-1))));
90
+ const key = candidates[0];
91
+ if (!key)
92
+ return false;
93
+ data[key].pop();
94
+ if (key === 'nodes') {
95
+ const ids = new Set(data.nodes.map((node) => node.id));
96
+ data.edges = data.edges.filter((edge) => ids.has(edge.from) && ids.has(edge.to));
97
+ data.frontier_remaining = (data.frontier_remaining ?? 0) + 1;
98
+ data.traversal_complete = false;
99
+ }
100
+ else if (key === 'edges') {
101
+ data.frontier_remaining = (data.frontier_remaining ?? 0) + 1;
102
+ data.traversal_complete = false;
103
+ }
104
+ return true;
105
+ }
106
+ /** Bound the complete structured response, not just result.data. Retained data
107
+ * continues to satisfy its shape, and retained evidence links stay resolvable. */
108
+ export function fitResponse(response, limits) {
109
+ const limit = responseByteLimit(limits);
110
+ if (measureResponse(response).metrics.returned_bytes <= limit)
111
+ return response;
112
+ markPartial(response);
113
+ response.warnings = [TRUNCATED, ...response.warnings.filter(item => item !== TRUNCATED).map(item => shorten(item, 256))];
114
+ pruneProvenance(response);
115
+ while (measureResponse(response).metrics.returned_bytes > limit) {
116
+ if (shortenLargestText(response.data))
117
+ continue;
118
+ if (removeResultEntry(response.data)) {
119
+ pruneProvenance(response);
120
+ continue;
121
+ }
122
+ if (response.warnings.length > 1) {
123
+ response.warnings = [TRUNCATED, 'Additional warnings omitted.'];
124
+ if (measureResponse(response).metrics.returned_bytes <= limit)
125
+ break;
126
+ response.warnings = [TRUNCATED];
127
+ continue;
128
+ }
129
+ // A long diagnostic is allowed to shrink, but error codes and rejection
130
+ // status survive; an exhausted error body becomes the bounded budget error.
131
+ if (response.error && response.error.message.length > 64) {
132
+ response.error.message = shorten(response.error.message, 63) + '…';
133
+ continue;
134
+ }
135
+ return budgetError(response);
136
+ }
137
+ return response;
138
+ }
@@ -5,6 +5,7 @@ export const StatusSchema = z.enum(['success', 'partial', 'blocked', 'error']);
5
5
  export const ResolutionSchema = z.enum(['resolved', 'unresolved', 'ambiguous', 'not_applicable']);
6
6
  export const FreshnessSchema = z.enum(['current', 'stale', 'unknown']);
7
7
  export const ProvenanceSchema = z.object({
8
+ evidence_id: z.string().min(1).optional(),
8
9
  backend: BackendNameSchema,
9
10
  version: z.string().min(1),
10
11
  source_path: z.string().nullable(),
@@ -26,6 +27,7 @@ export const MetricsSchema = z.object({
26
27
  latency_ms: z.number().nonnegative(),
27
28
  result_tokens: z.number().int().nonnegative(),
28
29
  token_count_kind: z.enum(['exact', 'estimated']),
30
+ token_count_method: z.literal('utf8_bytes_conservative').optional(),
29
31
  returned_bytes: z.number().int().nonnegative(),
30
32
  });
31
33
  export const ErrorSchema = z.object({