@hecer/yoke 1.21.1 → 1.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +48 -0
  4. package/README.md +8 -1
  5. package/TODOS.md +6 -0
  6. package/bench/analyze-codex-comparison.mjs +90 -17
  7. package/bench/compare-codex.mjs +159 -36
  8. package/bench/result-schema.mjs +132 -0
  9. package/canon/manifest.yaml +1 -1
  10. package/canon/skills/visual-verification/SKILL.md +25 -2
  11. package/canon/tools/codex-rtk-hook.mjs +6 -16
  12. package/dist/agents/pi-telemetry.js +2 -1
  13. package/dist/agents/process-streams.js +12 -64
  14. package/dist/agents/provider-selection.js +12 -0
  15. package/dist/agents/telemetry.js +52 -52
  16. package/dist/change/inbox.js +8 -3
  17. package/dist/check/command.js +69 -17
  18. package/dist/check/delivery.js +121 -0
  19. package/dist/cli.js +91 -3
  20. package/dist/code-intelligence/adapters/mcp.js +1 -0
  21. package/dist/code-intelligence/budgets.js +138 -0
  22. package/dist/code-intelligence/contracts.js +2 -0
  23. package/dist/code-intelligence/coordinator.js +159 -85
  24. package/dist/code-intelligence/evidence.js +87 -34
  25. package/dist/code-intelligence/index.js +1 -0
  26. package/dist/code-intelligence/mcp-client.js +25 -6
  27. package/dist/code-intelligence/mcp-server.js +14 -11
  28. package/dist/code-intelligence/preflight.js +71 -0
  29. package/dist/dashboard/analytics.js +5 -3
  30. package/dist/goals/command.js +183 -53
  31. package/dist/goals/usage.js +87 -0
  32. package/dist/loop/cache-isolation.js +36 -0
  33. package/dist/loop/candidate-cleanup.js +47 -17
  34. package/dist/loop/candidates.js +17 -11
  35. package/dist/loop/dispatcher.js +89 -26
  36. package/dist/loop/failure.js +104 -0
  37. package/dist/loop/gate-snapshot.js +19 -0
  38. package/dist/loop/git.js +1 -1
  39. package/dist/loop/loop.js +124 -70
  40. package/dist/loop/parallel-adapters.js +57 -6
  41. package/dist/loop/parallel-command.js +49 -7
  42. package/dist/loop/proof-retention.js +70 -0
  43. package/dist/loop/recovery.js +23 -5
  44. package/dist/loop/reporter.js +22 -5
  45. package/dist/loop/run-command.js +101 -47
  46. package/dist/loop/runner.js +6 -5
  47. package/dist/loop/worker.js +152 -91
  48. package/dist/observability/history.js +2 -1
  49. package/dist/observability/invocation.js +42 -0
  50. package/dist/observability/local-report.js +120 -0
  51. package/dist/observability/usage.js +19 -0
  52. package/dist/prd/command.js +20 -7
  53. package/dist/prd/decompose.js +5 -2
  54. package/dist/retrofit/config.js +29 -2
  55. package/dist/retrofit/gitignore.js +12 -0
  56. package/dist/retrofit/planners/codex.js +20 -20
  57. package/dist/routing/attempts.js +241 -0
  58. package/dist/routing/capability.js +13 -9
  59. package/dist/routing/optimization.js +73 -0
  60. package/dist/routing/registry.js +7 -1
  61. package/dist/routing/router.js +282 -127
  62. package/dist/setup/command.js +8 -2
  63. package/dist/smoke/command.js +387 -85
  64. package/dist/update/check.js +1 -1
  65. package/docs/BENCHMARK-MANIFEST.md +131 -0
  66. package/docs/CODE-INTELLIGENCE.md +43 -1
  67. package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
  68. package/docs/DELIVERY-JOURNEYS.md +206 -0
  69. package/docs/ECONOMIC-ROUTING.md +180 -0
  70. package/docs/GOALS.md +61 -4
  71. package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
  72. package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
  73. package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
  74. package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
  75. package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
  76. package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
  77. package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
  78. package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
  79. package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
  80. package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
  81. package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
  82. package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
  83. package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
  84. package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
  85. package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
  86. package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
  87. package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
  88. package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
  89. package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
  90. package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
  91. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
  92. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
  93. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
  94. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
  95. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
  96. package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
  97. package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
  98. package/docs/parallel-execution.md +37 -9
  99. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
  100. package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
  101. package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
  102. package/gemini-extension.json +1 -1
  103. package/package.json +1 -1
@@ -10,3 +10,135 @@ export function validateResult(result) {
10
10
  if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
11
11
  return result
12
12
  }
13
+
14
+ // Comparison manifests are separate from the historical individual-run schema above.
15
+ // A "verified" manifest describes comparable recorded conditions, not a causal or
16
+ // statistically conclusive performance result.
17
+ const record = value => value !== null && typeof value === 'object' && !Array.isArray(value)
18
+ const nonempty = value => typeof value === 'string' && value.trim().length > 0
19
+ const boolean = value => typeof value === 'boolean'
20
+ const sha256 = value => typeof value === 'string' && /^[a-f0-9]{64}$/.test(value)
21
+ const positiveInteger = value => Number.isSafeInteger(value) && value > 0
22
+
23
+ export const comparisonContextFields = {
24
+ 'source.version': nonempty,
25
+ 'source.commit': value => typeof value === 'string' && /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/.test(value),
26
+ 'source.dirty': boolean,
27
+ 'source.buildDigest': sha256,
28
+ 'source.stable': value => value === true,
29
+ 'fixture.id': nonempty,
30
+ 'fixture.seedDigest': sha256,
31
+ 'fixture.acceptanceDigest': sha256,
32
+ 'fixture.requirementsDigest': sha256,
33
+ 'execution.provider': nonempty,
34
+ 'execution.requestedModel': nonempty,
35
+ 'execution.actualModels': value => Array.isArray(value) && value.length > 0 && value.every(nonempty) && new Set(value).size === value.length,
36
+ 'execution.effort': nonempty,
37
+ 'execution.routing': boolean,
38
+ 'execution.nativeMultiAgent': boolean,
39
+ 'execution.nativeGoal': boolean,
40
+ 'execution.parallel': positiveInteger,
41
+ 'execution.workflow': nonempty,
42
+ 'execution.promptDigest': sha256,
43
+ 'execution.promptScope': value => ['provider-prompt', 'workflow-input-bundle'].includes(value),
44
+ 'startup.permissionProfile': value => ['safe', 'read-only', 'unsafe'].includes(value),
45
+ 'startup.bare': boolean,
46
+ 'startup.ignoreRules': boolean,
47
+ 'startup.commitPolicy': nonempty,
48
+ 'startup.isolation': nonempty,
49
+ 'startup.timeoutPolicy': nonempty,
50
+ 'startup.userStatePolicy': nonempty,
51
+ 'environment.platform': nonempty,
52
+ 'environment.arch': nonempty,
53
+ 'environment.nodeVersion': nonempty,
54
+ 'environment.providerVersion': nonempty,
55
+ 'environment.hostDigest': sha256,
56
+ 'environment.hostLoad': value => ['uncontrolled', 'isolated'].includes(value),
57
+ 'environment.skillsPlugins': value => ['not-audited', 'audited'].includes(value),
58
+ }
59
+
60
+ export const allowedComparisonVariables = Object.keys(comparisonContextFields)
61
+ .filter(field => !field.startsWith('fixture.') && field !== 'source.stable')
62
+
63
+ export function comparisonValue(context, field) {
64
+ return field.split('.').reduce((value, key) => record(value) ? value[key] : undefined, context)
65
+ }
66
+
67
+ export function stableComparisonJson(value) {
68
+ if (Array.isArray(value)) return '[' + value.map(stableComparisonJson).join(',') + ']'
69
+ if (record(value)) return '{' + Object.keys(value).sort().map(key => JSON.stringify(key) + ':' + stableComparisonJson(value[key])).join(',') + '}'
70
+ return JSON.stringify(value)
71
+ }
72
+
73
+ export function comparisonManifestIssues(manifest) {
74
+ if (!record(manifest)) return ['Missing versioned comparison manifest']
75
+ const issues = []
76
+ if (manifest.schemaVersion !== 1) issues.push('Unsupported comparison manifest version')
77
+ if (!nonempty(manifest.id)) issues.push('Missing comparison identity')
78
+ if (!['workflow', 'controlled'].includes(manifest.kind)) issues.push('Unknown comparison kind')
79
+ if (!Array.isArray(manifest.arms) || manifest.arms.length < 2 || !manifest.arms.every(nonempty) || new Set(manifest.arms).size !== manifest.arms.length) issues.push('Declare at least two distinct comparison arms')
80
+ if (!positiveInteger(manifest.repeats)) issues.push('Declare the planned number of paired repeats')
81
+ if (!Array.isArray(manifest.allowedDifferences)) issues.push('Declare allowed differences explicitly, including an empty list when appropriate')
82
+ else {
83
+ const fields = new Set()
84
+ for (const difference of manifest.allowedDifferences) {
85
+ if (!record(difference) || !allowedComparisonVariables.includes(difference.field) || !nonempty(difference.reason)) {
86
+ issues.push('Invalid comparison variable or missing rationale')
87
+ continue
88
+ }
89
+ if (fields.has(difference.field)) issues.push('Duplicate comparison variable: ' + difference.field)
90
+ fields.add(difference.field)
91
+ }
92
+ }
93
+ return issues
94
+ }
95
+
96
+ export function comparisonContextIssues(context) {
97
+ if (!record(context)) return ['Missing per-run comparison context']
98
+ const issues = []
99
+ for (const [field, validate] of Object.entries(comparisonContextFields)) {
100
+ if (!validate(comparisonValue(context, field))) issues.push('Missing, unknown or invalid condition: ' + field)
101
+ }
102
+ // Refuse silently ignored startup/source conditions introduced by a future writer.
103
+ for (const [section, fields] of Object.entries(context)) {
104
+ if (!record(fields)) { issues.push('Invalid context section: ' + section); continue }
105
+ for (const field of Object.keys(fields)) {
106
+ if (!(section + '.' + field in comparisonContextFields)) issues.push('Unknown comparison condition: ' + section + '.' + field)
107
+ }
108
+ }
109
+ return issues
110
+ }
111
+
112
+ export function assessComparison(manifests, runs) {
113
+ const manifest = manifests[0]
114
+ let reasons = manifests.flatMap(comparisonManifestIssues)
115
+ if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
116
+ if (manifests.some(value => stableComparisonJson(value) !== stableComparisonJson(manifest))) return { status: 'incompatible', reasons: ['Comparison manifests differ for the same identity'] }
117
+ reasons = runs.flatMap(run => comparisonContextIssues(run.context).map(issue => run.arm + ': ' + issue))
118
+ for (const run of runs) {
119
+ if (!manifest.arms.includes(run.arm) || !positiveInteger(run.repeat) || run.repeat > manifest.repeats) reasons.push('Unexpected arm or repeat')
120
+ if (run.fixture !== run.context?.fixture?.id) reasons.push('Fixture label disagrees with measured context')
121
+ if (typeof run.accepted !== 'boolean') reasons.push('Missing acceptance outcome')
122
+ }
123
+ if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
124
+ const keys = runs.map(run => run.arm + ':' + run.repeat)
125
+ if (new Set(keys).size !== keys.length) return { status: 'incompatible', reasons: ['Duplicate arm/repeat measurements'] }
126
+ const permitted = new Set(manifest.allowedDifferences.map(value => value.field))
127
+ for (const field of Object.keys(comparisonContextFields)) {
128
+ const valueOf = run => {
129
+ const value = comparisonValue(run.context, field)
130
+ return stableComparisonJson(field === 'execution.actualModels' ? [...value].sort() : value)
131
+ }
132
+ for (const arm of manifest.arms) {
133
+ if (new Set(runs.filter(run => run.arm === arm).map(valueOf)).size > 1) reasons.push('Conditions changed within arm ' + arm + ': ' + field)
134
+ }
135
+ if (!permitted.has(field) && new Set(runs.map(valueOf)).size > 1) reasons.push('Undeclared difference between arms: ' + field)
136
+ }
137
+ if (reasons.length) return { status: 'incompatible', reasons: [...new Set(reasons)] }
138
+ if (runs.length !== manifest.arms.length * manifest.repeats) return { status: 'incomplete', reasons: ['Not all declared paired runs are present'] }
139
+ if (runs.some(run => !run.accepted)) return { status: 'acceptance-failed', reasons: ['At least one arm failed immutable acceptance; failed elapsed time is not a speedup'] }
140
+ if (runs.some(run => run.usage?.measurementComplete === false || ['inputTokens', 'cachedInputTokens', 'outputTokens'].some(field => !Number.isFinite(run.usage?.[field]) || run.usage[field] < 0))) {
141
+ return { status: 'unverified', reasons: ['Incomplete token telemetry; retain diagnostic timings without an efficiency comparison'] }
142
+ }
143
+ return { status: 'verified', reasons: [] }
144
+ }
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.19.0
2
+ version: 1.23.0
3
3
  agents: [claude, codex, gemini, qwen, opencode, kilo, pi, hermes]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
@@ -24,6 +24,9 @@ Configure the key user flows once in `.yoke/config.yaml`:
24
24
  ```yaml
25
25
  smoke:
26
26
  baseUrl: http://localhost:3000
27
+ sourceIdentity:
28
+ path: /assets/app-specific-static-source.js
29
+ sha256: "<replace with SHA-256 of the served bytes>"
27
30
  flows:
28
31
  - name: home
29
32
  path: /
@@ -33,7 +36,26 @@ smoke:
33
36
  landmark: "form"
34
37
  ```
35
38
 
36
- With that in place, the `yoke flow-smoke .` step from the section-1 pipeline is live.
39
+ The example contains a required hash placeholder. Before running production smoke, replace
40
+ the resource path with a stable, app-specific source resource and the hash with its actual
41
+ 64-character lowercase SHA-256. The schema allows the field to be absent for existing
42
+ configuration readers; production `flow-smoke` requires it.
43
+
44
+ Start the intended server and confirm its checkout/build and port. Fetch the resource from
45
+ the effective `baseUrl` origin, using the exact body bytes returned by the server. Hash those
46
+ bytes, not a local source file that a dev server may transform. The resource must return a
47
+ successful HTTP response without redirects, finish within five seconds and contain at most
48
+ 1 MiB. Keep it stable for the whole smoke run. When `--url` changes the origin, verify that
49
+ origin serves the intended build before deriving the pin.
50
+
51
+ Refresh the expected hash explicitly after an intentional source/build change and after
52
+ confirming the server serves that change. A mismatch must fail the gate; do not automatically
53
+ overwrite the expected hash with whatever a port happens to serve. A shared health response
54
+ or unchanged marker cannot identify changed app code. This static pin proves only that the
55
+ selected resource matches before and after the run; retain the checkout fingerprint and
56
+ acceptance evidence for broader source binding.
57
+
58
+ With a real pin in place, the `yoke flow-smoke .` step from the section-1 pipeline is live.
37
59
  `yoke flow-smoke` loads each route against the running dev server, waits for the landmark,
38
60
  fails on any console error, and **always** saves a screenshot to `.yoke/proof/<story>/`
39
61
  (the loop labels the folder with the current story id via `YOKE_STORY`; standalone runs use
@@ -50,5 +72,6 @@ already handles the failure case.
50
72
 
51
73
  ## Rule
52
74
 
53
- Green pipeline = types + units + no design-slop over budget + every flow renders without
75
+ Green pipeline = types + units + no design-slop over budget + served-source identity matches
76
+ before and after the run + every flow renders without
54
77
  console errors, with a screenshot to prove it. Only then is the story actually done.
@@ -2,24 +2,14 @@ import { spawnSync } from 'node:child_process'
2
2
  import { resolve } from 'node:path'
3
3
  import { pathToFileURL } from 'node:url'
4
4
 
5
- function rtkCheck(command) {
6
- const result = spawnSync('rtk', ['hook', 'check', command], { encoding: 'utf8', timeout: 3000 })
7
- return result.status === 0 ? result.stdout.trim() : ''
5
+ function nativeHook(input) {
6
+ const result = spawnSync('rtk', ['hook', 'codex'], { input: JSON.stringify(input), encoding: 'utf8', timeout: 3000 })
7
+ return result.status === 0 && result.stdout.trim() ? JSON.parse(result.stdout) : null
8
8
  }
9
9
 
10
- export function rewriteHookInput(input, check = rtkCheck) {
11
- if (input?.tool_name !== 'Bash' && input?.toolName !== 'Bash') return null
12
- const toolInput = input.tool_input ?? input.toolInput
13
- const command = toolInput?.command
14
- if (typeof command !== 'string' || command.trim() === '') return null
15
- const rewritten = check(command)
16
- if (!rewritten || rewritten === command) return null
17
- return {
18
- hookSpecificOutput: {
19
- hookEventName: 'PreToolUse',
20
- updatedInput: { ...toolInput, command: rewritten },
21
- },
22
- }
10
+ export function rewriteHookInput(input, processHook = nativeHook) {
11
+ // Native RTK owns Codex schema, permission-mode handling and fail-open rules.
12
+ try { return processHook(input) ?? null } catch { return null }
23
13
  }
24
14
 
25
15
  async function main() {
@@ -39,10 +39,11 @@ export function createPiTelemetry() {
39
39
  if (complete) {
40
40
  // Optional totals must also cover every turn; omitted is not measured zero.
41
41
  const measured = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] === turns));
42
+ const partial = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] !== turns));
42
43
  return { usageAvailable: true, tokens: {
43
44
  ...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens,
44
45
  ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}),
45
- }, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
46
+ }, ...(Object.keys(partial).length ? { partialUsage: partial } : {}), ...(reportedModels.length > 1 ? { reportedModels } : {}) };
46
47
  }
47
48
  return { usageAvailable: false,
48
49
  ...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}),
@@ -1,4 +1,4 @@
1
- import { parseProviderTelemetry } from './telemetry.js';
1
+ import { createStepTelemetry, parseProviderTelemetry } from './telemetry.js';
2
2
  import { createPiTelemetry } from './pi-telemetry.js';
3
3
  export function createBoundedOutput(limitBytes) {
4
4
  let text = '';
@@ -22,11 +22,17 @@ export function createTelemetryAccumulator(agent) {
22
22
  let trailing = '';
23
23
  let telemetry = { usageAvailable: false };
24
24
  let reportedModels = [];
25
- const stepTotals = agent === 'opencode' || agent === 'kilo'
26
- ? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
27
- : undefined;
25
+ const steps = agent === 'opencode' || agent === 'kilo' ? createStepTelemetry() : undefined;
28
26
  const update = (lines) => {
29
27
  for (const line of lines) {
28
+ if (steps) {
29
+ try {
30
+ const event = JSON.parse(line);
31
+ if (event && typeof event === 'object' && !Array.isArray(event))
32
+ steps.consume(event);
33
+ }
34
+ catch { /* non-JSON diagnostics carry no usage */ }
35
+ }
30
36
  if (pi) {
31
37
  try {
32
38
  const event = JSON.parse(line);
@@ -45,35 +51,6 @@ export function createTelemetryAccumulator(agent) {
45
51
  // never add it to earlier results or to assistant-message snapshots.
46
52
  if (next.tokens || next.partialUsage)
47
53
  telemetry = next;
48
- if (stepTotals && isStepFinish(line)) {
49
- const usage = next.tokens ?? next.partialUsage;
50
- if (usage) {
51
- if (usage.inputTokens !== undefined) {
52
- stepTotals.input += usage.inputTokens;
53
- stepTotals.hasInput = true;
54
- }
55
- if (usage.outputTokens !== undefined) {
56
- stepTotals.output += usage.outputTokens;
57
- stepTotals.hasOutput = true;
58
- }
59
- if (usage.cachedInputTokens !== undefined) {
60
- stepTotals.cached += usage.cachedInputTokens;
61
- stepTotals.hasCached = true;
62
- }
63
- if (usage.cacheWriteInputTokens !== undefined) {
64
- stepTotals.cacheWrite += usage.cacheWriteInputTokens;
65
- stepTotals.hasCacheWrite = true;
66
- }
67
- if (usage.reasoningOutputTokens !== undefined) {
68
- stepTotals.reasoning += usage.reasoningOutputTokens;
69
- stepTotals.hasReasoning = true;
70
- }
71
- if (usage.totalCostUsd !== undefined) {
72
- stepTotals.cost += usage.totalCostUsd;
73
- stepTotals.hasCost = true;
74
- }
75
- }
76
- }
77
54
  }
78
55
  };
79
56
  return {
@@ -88,42 +65,13 @@ export function createTelemetryAccumulator(agent) {
88
65
  trailing = '';
89
66
  if (pi)
90
67
  return pi.finish();
91
- if (stepTotals && (stepTotals.hasInput || stepTotals.hasOutput)) {
92
- const latest = telemetry.tokens;
93
- const inputTokens = stepTotals.hasInput ? stepTotals.input : latest?.inputTokens;
94
- const outputTokens = stepTotals.hasOutput ? stepTotals.output : latest?.outputTokens;
95
- const partialUsage = {
96
- ...(inputTokens !== undefined ? { inputTokens } : {}),
97
- ...(outputTokens !== undefined ? { outputTokens } : {}),
98
- ...(stepTotals.hasCached ? { cachedInputTokens: stepTotals.cached } : latest?.cachedInputTokens !== undefined ? { cachedInputTokens: latest.cachedInputTokens } : {}),
99
- ...(stepTotals.hasCacheWrite ? { cacheWriteInputTokens: stepTotals.cacheWrite } : latest?.cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens: latest.cacheWriteInputTokens } : {}),
100
- ...(stepTotals.hasReasoning ? { reasoningOutputTokens: stepTotals.reasoning } : latest?.reasoningOutputTokens !== undefined ? { reasoningOutputTokens: latest.reasoningOutputTokens } : {}),
101
- ...(stepTotals.hasCost ? { totalCostUsd: stepTotals.cost } : latest?.totalCostUsd !== undefined ? { totalCostUsd: latest.totalCostUsd } : {}),
102
- ...(latest?.model ? { model: latest.model } : {}),
103
- };
104
- if (inputTokens !== undefined && outputTokens !== undefined) {
105
- telemetry = { usageAvailable: true, tokens: { ...partialUsage, inputTokens, outputTokens } };
106
- }
107
- else {
108
- telemetry = { usageAvailable: false, partialUsage };
109
- }
110
- }
68
+ telemetry = steps?.finish() ?? telemetry;
111
69
  if (telemetry.tokens) {
112
70
  const { model: _model, ...tokens } = telemetry.tokens;
113
- return { usageAvailable: telemetry.usageAvailable, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
71
+ return { ...telemetry, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
114
72
  ...(reportedModels.length > 1 ? { reportedModels } : {}) };
115
73
  }
116
74
  return { ...telemetry, ...(reportedModels.length ? { reportedModels } : {}) };
117
75
  },
118
76
  };
119
77
  }
120
- function isStepFinish(line) {
121
- try {
122
- const value = JSON.parse(line);
123
- const part = value.part && typeof value.part === 'object' ? value.part : undefined;
124
- return value.type === 'step_finish' || part?.type === 'step-finish';
125
- }
126
- catch {
127
- return false;
128
- }
129
- }
@@ -0,0 +1,12 @@
1
+ /** Runner defaults belong to the runner that selected them. Undefined overrides
2
+ * are absent; explicit false values (for example bare: false) still override. */
3
+ export function resolveProviderSelection(agent, defaults, overrides = {}) {
4
+ const inherited = defaults && (defaults.agent ?? 'codex') === agent ? defaults : {};
5
+ const result = {};
6
+ for (const key of ['provider', 'model', 'reasoningEffort', 'variant', 'bare', 'nativeMultiAgent']) {
7
+ const value = overrides[key] ?? inherited[key];
8
+ if (value !== undefined)
9
+ Object.assign(result, { [key]: value });
10
+ }
11
+ return result;
12
+ }
@@ -13,6 +13,50 @@ function parseJson(value) {
13
13
  function isRecord(value) {
14
14
  return typeof value === 'object' && value !== null && !Array.isArray(value);
15
15
  }
16
+ /** OpenCode-family usage belongs to individual completed steps. Each field needs
17
+ * coverage of every step before it can be called a total. Shared by batch and
18
+ * streaming readers so an omitted measurement cannot turn into a measured zero. */
19
+ export function createStepTelemetry() {
20
+ let steps = 0;
21
+ const totals = {};
22
+ const counts = {};
23
+ return {
24
+ consume(event) {
25
+ const part = isRecord(event.part) ? event.part : undefined;
26
+ if (event.type !== 'step_finish' && part?.type !== 'step-finish')
27
+ return;
28
+ steps++;
29
+ const tokens = isRecord(part?.tokens) ? part.tokens : {};
30
+ const cache = isRecord(tokens.cache) ? tokens.cache : undefined;
31
+ const fields = {
32
+ inputTokens: tokens.input,
33
+ outputTokens: tokens.output,
34
+ cachedInputTokens: tokens.cached ?? tokens.cacheRead ?? cache?.read,
35
+ cacheWriteInputTokens: tokens.cacheWrite ?? cache?.write,
36
+ reasoningOutputTokens: tokens.reasoning,
37
+ totalCostUsd: part?.cost ?? event.cost,
38
+ };
39
+ for (const [field, raw] of Object.entries(fields)) {
40
+ const value = finite(raw);
41
+ if (value === undefined)
42
+ continue;
43
+ totals[field] = (totals[field] ?? 0) + value;
44
+ counts[field] = (counts[field] ?? 0) + 1;
45
+ }
46
+ },
47
+ finish() {
48
+ if (!steps)
49
+ return undefined;
50
+ const complete = counts.inputTokens === steps && counts.outputTokens === steps;
51
+ if (!complete)
52
+ return { usageAvailable: false, ...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}) };
53
+ const measured = Object.fromEntries(Object.entries(totals).filter(([field]) => counts[field] === steps));
54
+ const partialUsage = Object.fromEntries(Object.entries(totals).filter(([field]) => counts[field] !== steps));
55
+ return { usageAvailable: true, tokens: { ...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens },
56
+ ...(Object.keys(partialUsage).length ? { partialUsage } : {}) };
57
+ },
58
+ };
59
+ }
16
60
  function textContent(value) {
17
61
  if (typeof value === 'string')
18
62
  return value;
@@ -154,9 +198,7 @@ export function parseProviderTelemetry(agent, lines) {
154
198
  let totalCostUsd;
155
199
  let model;
156
200
  let reportedModels = [];
157
- const harnessTotals = agent === 'opencode' || agent === 'kilo'
158
- ? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
159
- : undefined;
201
+ const steps = agent === 'opencode' || agent === 'kilo' ? createStepTelemetry() : undefined;
160
202
  for (const line of lines) {
161
203
  let parsed;
162
204
  try {
@@ -168,45 +210,11 @@ export function parseProviderTelemetry(agent, lines) {
168
210
  if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed))
169
211
  continue;
170
212
  const event = parsed;
213
+ steps?.consume(event);
171
214
  if ((agent === 'qwen' || agent === 'claude') && event.parent_tool_use_id != null)
172
215
  continue;
173
216
  const message = event.message && typeof event.message === 'object' ? event.message : undefined;
174
217
  const stats = event.stats && typeof event.stats === 'object' ? event.stats : undefined;
175
- const part = event.part && typeof event.part === 'object' ? event.part : undefined;
176
- const partTokens = part?.tokens && typeof part.tokens === 'object' ? part.tokens : undefined;
177
- if (harnessTotals && (event.type === 'step_finish' || part?.type === 'step-finish') && partTokens) {
178
- const cache = partTokens.cache && typeof partTokens.cache === 'object' ? partTokens.cache : undefined;
179
- const stepInput = finite(partTokens.input);
180
- const stepOutput = finite(partTokens.output);
181
- const stepCached = finite(partTokens.cached ?? partTokens.cacheRead ?? cache?.read);
182
- const stepCacheWrite = finite(partTokens.cacheWrite ?? cache?.write);
183
- const stepReasoning = finite(partTokens.reasoning);
184
- const stepCost = finite(part?.cost ?? event.cost);
185
- if (stepInput !== undefined) {
186
- harnessTotals.input += stepInput;
187
- harnessTotals.hasInput = true;
188
- }
189
- if (stepOutput !== undefined) {
190
- harnessTotals.output += stepOutput;
191
- harnessTotals.hasOutput = true;
192
- }
193
- if (stepCached !== undefined) {
194
- harnessTotals.cached += stepCached;
195
- harnessTotals.hasCached = true;
196
- }
197
- if (stepCacheWrite !== undefined) {
198
- harnessTotals.cacheWrite += stepCacheWrite;
199
- harnessTotals.hasCacheWrite = true;
200
- }
201
- if (stepReasoning !== undefined) {
202
- harnessTotals.reasoning += stepReasoning;
203
- harnessTotals.hasReasoning = true;
204
- }
205
- if (stepCost !== undefined) {
206
- harnessTotals.cost += stepCost;
207
- harnessTotals.hasCost = true;
208
- }
209
- }
210
218
  const usage = (event.usage && typeof event.usage === 'object'
211
219
  ? event.usage
212
220
  : message?.usage && typeof message.usage === 'object'
@@ -277,20 +285,12 @@ export function parseProviderTelemetry(agent, lines) {
277
285
  if (typeof eventModel === 'string' && eventModel && reportedModels.length <= 1)
278
286
  model = eventModel;
279
287
  }
280
- if (harnessTotals) {
281
- if (harnessTotals.hasInput)
282
- inputTokens = harnessTotals.input;
283
- if (harnessTotals.hasOutput)
284
- outputTokens = harnessTotals.output;
285
- if (harnessTotals.hasCached)
286
- cachedInputTokens = harnessTotals.cached;
287
- if (harnessTotals.hasCacheWrite)
288
- cacheWriteInputTokens = harnessTotals.cacheWrite;
289
- if (harnessTotals.hasReasoning)
290
- reasoningOutputTokens = harnessTotals.reasoning;
291
- if (harnessTotals.hasCost)
292
- totalCostUsd = harnessTotals.cost;
293
- }
288
+ const stepUsage = steps?.finish();
289
+ if (stepUsage)
290
+ return { ...stepUsage,
291
+ ...(stepUsage.tokens && model ? { tokens: { ...stepUsage.tokens, model } } : {}),
292
+ ...(!stepUsage.tokens && model ? { reportedModels: [model] } : {}),
293
+ };
294
294
  if (inputTokens === undefined || outputTokens === undefined) {
295
295
  const partialUsage = {
296
296
  ...(inputTokens !== undefined ? { inputTokens } : {}),
@@ -9,8 +9,9 @@ import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, w
9
9
  import { basename, dirname, isAbsolute, join, resolve } from 'node:path';
10
10
  import { z } from 'zod';
11
11
  import { criterionCommandProblem, isAcceptanceCriterion, loadPrd, parsePrd, savePrd, validateDependencies } from '../loop/prd.js';
12
- import { agentInvocation, buildWatchdogInvocation, isAgentAvailable, runAgent, } from '../loop/runner.js';
12
+ import { agentInvocation, buildWatchdogInvocation, isAgentAvailable, runCapturedAgent, } from '../loop/runner.js';
13
13
  import { commitPaths } from '../loop/git.js';
14
+ import { measureInvocation } from '../observability/invocation.js';
14
15
  const ChangeRequestSchema = z.object({
15
16
  version: z.literal(1),
16
17
  id: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/),
@@ -184,7 +185,9 @@ export function runChangeApply(targetDir, opts) {
184
185
  rmSync(proposal, { force: true });
185
186
  const base = agentInvocation(opts.runner, buildChangePrompt(request, proposal, existing, brief), plannerDir, opts.permissions ?? 'safe', opts.selection);
186
187
  const invocation = buildWatchdogInvocation(base, opts.timeoutMs ?? 0);
187
- const result = (opts.run ?? runAgent)(invocation);
188
+ const result = measureInvocation({ root: targetDir, agent: opts.runner, role: 'change-planner', storyId: request.id, runId: `change:${request.id}`, selection: opts.selection, invocation,
189
+ execute: opts.run ?? (item => runCapturedAgent(opts.runner, item)),
190
+ });
188
191
  if (!result.success)
189
192
  return { ok: false, added: 0, summary: `planner failed: ${result.summary}`, changeId: request.id };
190
193
  if (readFileSync(prdPath, 'utf8') !== existingText || (readPlanningFile(targetDir, '.yoke/plan.md', 80_000) ?? '') !== brief) {
@@ -246,7 +249,9 @@ export function runChangeApply(targetDir, opts) {
246
249
  rmSync(reviewPath, { force: true });
247
250
  const reviewBase = agentInvocation(reviewer, buildChangeReviewPrompt(request, reviewPath, appended), plannerDir, opts.permissions ?? 'safe', reviewer === opts.runner ? opts.selection : undefined);
248
251
  const reviewInvocation = buildWatchdogInvocation(reviewBase, opts.timeoutMs ?? 0);
249
- const reviewResult = (opts.review ?? runAgent)(reviewInvocation);
252
+ const reviewResult = measureInvocation({ root: targetDir, agent: reviewer, role: 'coverage-review', storyId: request.id, runId: `change:${request.id}`, selection: reviewer === opts.runner ? opts.selection : undefined, invocation: reviewInvocation,
253
+ execute: opts.review ?? (item => runCapturedAgent(reviewer, item)),
254
+ });
250
255
  if (!reviewResult.success) {
251
256
  return { ok: false, added: 0, summary: `coverage review failed: ${reviewResult.summary}`, changeId: request.id };
252
257
  }