@hecer/yoke 1.21.1 → 1.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +48 -0
- package/README.md +8 -1
- package/TODOS.md +6 -0
- package/bench/analyze-codex-comparison.mjs +90 -17
- package/bench/compare-codex.mjs +159 -36
- package/bench/result-schema.mjs +132 -0
- package/canon/manifest.yaml +1 -1
- package/canon/skills/visual-verification/SKILL.md +25 -2
- package/canon/tools/codex-rtk-hook.mjs +6 -16
- package/dist/agents/pi-telemetry.js +2 -1
- package/dist/agents/process-streams.js +12 -64
- package/dist/agents/provider-selection.js +12 -0
- package/dist/agents/telemetry.js +52 -52
- package/dist/change/inbox.js +8 -3
- package/dist/check/command.js +69 -17
- package/dist/check/delivery.js +121 -0
- package/dist/cli.js +91 -3
- package/dist/code-intelligence/adapters/mcp.js +1 -0
- package/dist/code-intelligence/budgets.js +138 -0
- package/dist/code-intelligence/contracts.js +2 -0
- package/dist/code-intelligence/coordinator.js +159 -85
- package/dist/code-intelligence/evidence.js +87 -34
- package/dist/code-intelligence/index.js +1 -0
- package/dist/code-intelligence/mcp-client.js +25 -6
- package/dist/code-intelligence/mcp-server.js +14 -11
- package/dist/code-intelligence/preflight.js +71 -0
- package/dist/dashboard/analytics.js +5 -3
- package/dist/goals/command.js +183 -53
- package/dist/goals/usage.js +87 -0
- package/dist/loop/cache-isolation.js +36 -0
- package/dist/loop/candidate-cleanup.js +47 -17
- package/dist/loop/candidates.js +17 -11
- package/dist/loop/dispatcher.js +89 -26
- package/dist/loop/failure.js +104 -0
- package/dist/loop/gate-snapshot.js +19 -0
- package/dist/loop/git.js +1 -1
- package/dist/loop/loop.js +124 -70
- package/dist/loop/parallel-adapters.js +57 -6
- package/dist/loop/parallel-command.js +49 -7
- package/dist/loop/proof-retention.js +70 -0
- package/dist/loop/recovery.js +23 -5
- package/dist/loop/reporter.js +22 -5
- package/dist/loop/run-command.js +101 -47
- package/dist/loop/runner.js +6 -5
- package/dist/loop/worker.js +152 -91
- package/dist/observability/history.js +2 -1
- package/dist/observability/invocation.js +42 -0
- package/dist/observability/local-report.js +120 -0
- package/dist/observability/usage.js +19 -0
- package/dist/prd/command.js +20 -7
- package/dist/prd/decompose.js +5 -2
- package/dist/retrofit/config.js +29 -2
- package/dist/retrofit/gitignore.js +12 -0
- package/dist/retrofit/planners/codex.js +20 -20
- package/dist/routing/attempts.js +241 -0
- package/dist/routing/capability.js +13 -9
- package/dist/routing/optimization.js +73 -0
- package/dist/routing/registry.js +7 -1
- package/dist/routing/router.js +282 -127
- package/dist/setup/command.js +8 -2
- package/dist/smoke/command.js +387 -85
- package/dist/update/check.js +1 -1
- package/docs/BENCHMARK-MANIFEST.md +131 -0
- package/docs/CODE-INTELLIGENCE.md +43 -1
- package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
- package/docs/DELIVERY-JOURNEYS.md +206 -0
- package/docs/ECONOMIC-ROUTING.md +180 -0
- package/docs/GOALS.md +61 -4
- package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
- package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
- package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
- package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
- package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
- package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
- package/docs/parallel-execution.md +37 -9
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
- package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
package/bench/result-schema.mjs
CHANGED
|
@@ -10,3 +10,135 @@ export function validateResult(result) {
|
|
|
10
10
|
if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
|
|
11
11
|
return result
|
|
12
12
|
}
|
|
13
|
+
|
|
14
|
+
// Comparison manifests are separate from the historical individual-run schema above.
|
|
15
|
+
// A "verified" manifest describes comparable recorded conditions, not a causal or
|
|
16
|
+
// statistically conclusive performance result.
|
|
17
|
+
const record = value => value !== null && typeof value === 'object' && !Array.isArray(value)
|
|
18
|
+
const nonempty = value => typeof value === 'string' && value.trim().length > 0
|
|
19
|
+
const boolean = value => typeof value === 'boolean'
|
|
20
|
+
const sha256 = value => typeof value === 'string' && /^[a-f0-9]{64}$/.test(value)
|
|
21
|
+
const positiveInteger = value => Number.isSafeInteger(value) && value > 0
|
|
22
|
+
|
|
23
|
+
export const comparisonContextFields = {
|
|
24
|
+
'source.version': nonempty,
|
|
25
|
+
'source.commit': value => typeof value === 'string' && /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/.test(value),
|
|
26
|
+
'source.dirty': boolean,
|
|
27
|
+
'source.buildDigest': sha256,
|
|
28
|
+
'source.stable': value => value === true,
|
|
29
|
+
'fixture.id': nonempty,
|
|
30
|
+
'fixture.seedDigest': sha256,
|
|
31
|
+
'fixture.acceptanceDigest': sha256,
|
|
32
|
+
'fixture.requirementsDigest': sha256,
|
|
33
|
+
'execution.provider': nonempty,
|
|
34
|
+
'execution.requestedModel': nonempty,
|
|
35
|
+
'execution.actualModels': value => Array.isArray(value) && value.length > 0 && value.every(nonempty) && new Set(value).size === value.length,
|
|
36
|
+
'execution.effort': nonempty,
|
|
37
|
+
'execution.routing': boolean,
|
|
38
|
+
'execution.nativeMultiAgent': boolean,
|
|
39
|
+
'execution.nativeGoal': boolean,
|
|
40
|
+
'execution.parallel': positiveInteger,
|
|
41
|
+
'execution.workflow': nonempty,
|
|
42
|
+
'execution.promptDigest': sha256,
|
|
43
|
+
'execution.promptScope': value => ['provider-prompt', 'workflow-input-bundle'].includes(value),
|
|
44
|
+
'startup.permissionProfile': value => ['safe', 'read-only', 'unsafe'].includes(value),
|
|
45
|
+
'startup.bare': boolean,
|
|
46
|
+
'startup.ignoreRules': boolean,
|
|
47
|
+
'startup.commitPolicy': nonempty,
|
|
48
|
+
'startup.isolation': nonempty,
|
|
49
|
+
'startup.timeoutPolicy': nonempty,
|
|
50
|
+
'startup.userStatePolicy': nonempty,
|
|
51
|
+
'environment.platform': nonempty,
|
|
52
|
+
'environment.arch': nonempty,
|
|
53
|
+
'environment.nodeVersion': nonempty,
|
|
54
|
+
'environment.providerVersion': nonempty,
|
|
55
|
+
'environment.hostDigest': sha256,
|
|
56
|
+
'environment.hostLoad': value => ['uncontrolled', 'isolated'].includes(value),
|
|
57
|
+
'environment.skillsPlugins': value => ['not-audited', 'audited'].includes(value),
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export const allowedComparisonVariables = Object.keys(comparisonContextFields)
|
|
61
|
+
.filter(field => !field.startsWith('fixture.') && field !== 'source.stable')
|
|
62
|
+
|
|
63
|
+
export function comparisonValue(context, field) {
|
|
64
|
+
return field.split('.').reduce((value, key) => record(value) ? value[key] : undefined, context)
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export function stableComparisonJson(value) {
|
|
68
|
+
if (Array.isArray(value)) return '[' + value.map(stableComparisonJson).join(',') + ']'
|
|
69
|
+
if (record(value)) return '{' + Object.keys(value).sort().map(key => JSON.stringify(key) + ':' + stableComparisonJson(value[key])).join(',') + '}'
|
|
70
|
+
return JSON.stringify(value)
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export function comparisonManifestIssues(manifest) {
|
|
74
|
+
if (!record(manifest)) return ['Missing versioned comparison manifest']
|
|
75
|
+
const issues = []
|
|
76
|
+
if (manifest.schemaVersion !== 1) issues.push('Unsupported comparison manifest version')
|
|
77
|
+
if (!nonempty(manifest.id)) issues.push('Missing comparison identity')
|
|
78
|
+
if (!['workflow', 'controlled'].includes(manifest.kind)) issues.push('Unknown comparison kind')
|
|
79
|
+
if (!Array.isArray(manifest.arms) || manifest.arms.length < 2 || !manifest.arms.every(nonempty) || new Set(manifest.arms).size !== manifest.arms.length) issues.push('Declare at least two distinct comparison arms')
|
|
80
|
+
if (!positiveInteger(manifest.repeats)) issues.push('Declare the planned number of paired repeats')
|
|
81
|
+
if (!Array.isArray(manifest.allowedDifferences)) issues.push('Declare allowed differences explicitly, including an empty list when appropriate')
|
|
82
|
+
else {
|
|
83
|
+
const fields = new Set()
|
|
84
|
+
for (const difference of manifest.allowedDifferences) {
|
|
85
|
+
if (!record(difference) || !allowedComparisonVariables.includes(difference.field) || !nonempty(difference.reason)) {
|
|
86
|
+
issues.push('Invalid comparison variable or missing rationale')
|
|
87
|
+
continue
|
|
88
|
+
}
|
|
89
|
+
if (fields.has(difference.field)) issues.push('Duplicate comparison variable: ' + difference.field)
|
|
90
|
+
fields.add(difference.field)
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
return issues
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export function comparisonContextIssues(context) {
|
|
97
|
+
if (!record(context)) return ['Missing per-run comparison context']
|
|
98
|
+
const issues = []
|
|
99
|
+
for (const [field, validate] of Object.entries(comparisonContextFields)) {
|
|
100
|
+
if (!validate(comparisonValue(context, field))) issues.push('Missing, unknown or invalid condition: ' + field)
|
|
101
|
+
}
|
|
102
|
+
// Refuse silently ignored startup/source conditions introduced by a future writer.
|
|
103
|
+
for (const [section, fields] of Object.entries(context)) {
|
|
104
|
+
if (!record(fields)) { issues.push('Invalid context section: ' + section); continue }
|
|
105
|
+
for (const field of Object.keys(fields)) {
|
|
106
|
+
if (!(section + '.' + field in comparisonContextFields)) issues.push('Unknown comparison condition: ' + section + '.' + field)
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
return issues
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export function assessComparison(manifests, runs) {
|
|
113
|
+
const manifest = manifests[0]
|
|
114
|
+
let reasons = manifests.flatMap(comparisonManifestIssues)
|
|
115
|
+
if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
|
|
116
|
+
if (manifests.some(value => stableComparisonJson(value) !== stableComparisonJson(manifest))) return { status: 'incompatible', reasons: ['Comparison manifests differ for the same identity'] }
|
|
117
|
+
reasons = runs.flatMap(run => comparisonContextIssues(run.context).map(issue => run.arm + ': ' + issue))
|
|
118
|
+
for (const run of runs) {
|
|
119
|
+
if (!manifest.arms.includes(run.arm) || !positiveInteger(run.repeat) || run.repeat > manifest.repeats) reasons.push('Unexpected arm or repeat')
|
|
120
|
+
if (run.fixture !== run.context?.fixture?.id) reasons.push('Fixture label disagrees with measured context')
|
|
121
|
+
if (typeof run.accepted !== 'boolean') reasons.push('Missing acceptance outcome')
|
|
122
|
+
}
|
|
123
|
+
if (reasons.length) return { status: 'unverified', reasons: [...new Set(reasons)] }
|
|
124
|
+
const keys = runs.map(run => run.arm + ':' + run.repeat)
|
|
125
|
+
if (new Set(keys).size !== keys.length) return { status: 'incompatible', reasons: ['Duplicate arm/repeat measurements'] }
|
|
126
|
+
const permitted = new Set(manifest.allowedDifferences.map(value => value.field))
|
|
127
|
+
for (const field of Object.keys(comparisonContextFields)) {
|
|
128
|
+
const valueOf = run => {
|
|
129
|
+
const value = comparisonValue(run.context, field)
|
|
130
|
+
return stableComparisonJson(field === 'execution.actualModels' ? [...value].sort() : value)
|
|
131
|
+
}
|
|
132
|
+
for (const arm of manifest.arms) {
|
|
133
|
+
if (new Set(runs.filter(run => run.arm === arm).map(valueOf)).size > 1) reasons.push('Conditions changed within arm ' + arm + ': ' + field)
|
|
134
|
+
}
|
|
135
|
+
if (!permitted.has(field) && new Set(runs.map(valueOf)).size > 1) reasons.push('Undeclared difference between arms: ' + field)
|
|
136
|
+
}
|
|
137
|
+
if (reasons.length) return { status: 'incompatible', reasons: [...new Set(reasons)] }
|
|
138
|
+
if (runs.length !== manifest.arms.length * manifest.repeats) return { status: 'incomplete', reasons: ['Not all declared paired runs are present'] }
|
|
139
|
+
if (runs.some(run => !run.accepted)) return { status: 'acceptance-failed', reasons: ['At least one arm failed immutable acceptance; failed elapsed time is not a speedup'] }
|
|
140
|
+
if (runs.some(run => run.usage?.measurementComplete === false || ['inputTokens', 'cachedInputTokens', 'outputTokens'].some(field => !Number.isFinite(run.usage?.[field]) || run.usage[field] < 0))) {
|
|
141
|
+
return { status: 'unverified', reasons: ['Incomplete token telemetry; retain diagnostic timings without an efficiency comparison'] }
|
|
142
|
+
}
|
|
143
|
+
return { status: 'verified', reasons: [] }
|
|
144
|
+
}
|
package/canon/manifest.yaml
CHANGED
|
@@ -24,6 +24,9 @@ Configure the key user flows once in `.yoke/config.yaml`:
|
|
|
24
24
|
```yaml
|
|
25
25
|
smoke:
|
|
26
26
|
baseUrl: http://localhost:3000
|
|
27
|
+
sourceIdentity:
|
|
28
|
+
path: /assets/app-specific-static-source.js
|
|
29
|
+
sha256: "<replace with SHA-256 of the served bytes>"
|
|
27
30
|
flows:
|
|
28
31
|
- name: home
|
|
29
32
|
path: /
|
|
@@ -33,7 +36,26 @@ smoke:
|
|
|
33
36
|
landmark: "form"
|
|
34
37
|
```
|
|
35
38
|
|
|
36
|
-
|
|
39
|
+
The example contains a required hash placeholder. Before running production smoke, replace
|
|
40
|
+
the resource path with a stable, app-specific source resource and the hash with its actual
|
|
41
|
+
64-character lowercase SHA-256. The schema allows the field to be absent for existing
|
|
42
|
+
configuration readers; production `flow-smoke` requires it.
|
|
43
|
+
|
|
44
|
+
Start the intended server and confirm its checkout/build and port. Fetch the resource from
|
|
45
|
+
the effective `baseUrl` origin, using the exact body bytes returned by the server. Hash those
|
|
46
|
+
bytes, not a local source file that a dev server may transform. The resource must return a
|
|
47
|
+
successful HTTP response without redirects, finish within five seconds and contain at most
|
|
48
|
+
1 MiB. Keep it stable for the whole smoke run. When `--url` changes the origin, verify that
|
|
49
|
+
origin serves the intended build before deriving the pin.
|
|
50
|
+
|
|
51
|
+
Refresh the expected hash explicitly after an intentional source/build change and after
|
|
52
|
+
confirming the server serves that change. A mismatch must fail the gate; do not automatically
|
|
53
|
+
overwrite the expected hash with whatever a port happens to serve. A shared health response
|
|
54
|
+
or unchanged marker cannot identify changed app code. This static pin proves only that the
|
|
55
|
+
selected resource matches before and after the run; retain the checkout fingerprint and
|
|
56
|
+
acceptance evidence for broader source binding.
|
|
57
|
+
|
|
58
|
+
With a real pin in place, the `yoke flow-smoke .` step from the section-1 pipeline is live.
|
|
37
59
|
`yoke flow-smoke` loads each route against the running dev server, waits for the landmark,
|
|
38
60
|
fails on any console error, and **always** saves a screenshot to `.yoke/proof/<story>/`
|
|
39
61
|
(the loop labels the folder with the current story id via `YOKE_STORY`; standalone runs use
|
|
@@ -50,5 +72,6 @@ already handles the failure case.
|
|
|
50
72
|
|
|
51
73
|
## Rule
|
|
52
74
|
|
|
53
|
-
Green pipeline = types + units + no design-slop over budget +
|
|
75
|
+
Green pipeline = types + units + no design-slop over budget + served-source identity matches
|
|
76
|
+
before and after the run + every flow renders without
|
|
54
77
|
console errors, with a screenshot to prove it. Only then is the story actually done.
|
|
@@ -2,24 +2,14 @@ import { spawnSync } from 'node:child_process'
|
|
|
2
2
|
import { resolve } from 'node:path'
|
|
3
3
|
import { pathToFileURL } from 'node:url'
|
|
4
4
|
|
|
5
|
-
function
|
|
6
|
-
const result = spawnSync('rtk', ['hook', '
|
|
7
|
-
return result.status === 0
|
|
5
|
+
function nativeHook(input) {
|
|
6
|
+
const result = spawnSync('rtk', ['hook', 'codex'], { input: JSON.stringify(input), encoding: 'utf8', timeout: 3000 })
|
|
7
|
+
return result.status === 0 && result.stdout.trim() ? JSON.parse(result.stdout) : null
|
|
8
8
|
}
|
|
9
9
|
|
|
10
|
-
export function rewriteHookInput(input,
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
const command = toolInput?.command
|
|
14
|
-
if (typeof command !== 'string' || command.trim() === '') return null
|
|
15
|
-
const rewritten = check(command)
|
|
16
|
-
if (!rewritten || rewritten === command) return null
|
|
17
|
-
return {
|
|
18
|
-
hookSpecificOutput: {
|
|
19
|
-
hookEventName: 'PreToolUse',
|
|
20
|
-
updatedInput: { ...toolInput, command: rewritten },
|
|
21
|
-
},
|
|
22
|
-
}
|
|
10
|
+
export function rewriteHookInput(input, processHook = nativeHook) {
|
|
11
|
+
// Native RTK owns Codex schema, permission-mode handling and fail-open rules.
|
|
12
|
+
try { return processHook(input) ?? null } catch { return null }
|
|
23
13
|
}
|
|
24
14
|
|
|
25
15
|
async function main() {
|
|
@@ -39,10 +39,11 @@ export function createPiTelemetry() {
|
|
|
39
39
|
if (complete) {
|
|
40
40
|
// Optional totals must also cover every turn; omitted is not measured zero.
|
|
41
41
|
const measured = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] === turns));
|
|
42
|
+
const partial = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] !== turns));
|
|
42
43
|
return { usageAvailable: true, tokens: {
|
|
43
44
|
...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens,
|
|
44
45
|
...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}),
|
|
45
|
-
}, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
|
|
46
|
+
}, ...(Object.keys(partial).length ? { partialUsage: partial } : {}), ...(reportedModels.length > 1 ? { reportedModels } : {}) };
|
|
46
47
|
}
|
|
47
48
|
return { usageAvailable: false,
|
|
48
49
|
...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}),
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { parseProviderTelemetry } from './telemetry.js';
|
|
1
|
+
import { createStepTelemetry, parseProviderTelemetry } from './telemetry.js';
|
|
2
2
|
import { createPiTelemetry } from './pi-telemetry.js';
|
|
3
3
|
export function createBoundedOutput(limitBytes) {
|
|
4
4
|
let text = '';
|
|
@@ -22,11 +22,17 @@ export function createTelemetryAccumulator(agent) {
|
|
|
22
22
|
let trailing = '';
|
|
23
23
|
let telemetry = { usageAvailable: false };
|
|
24
24
|
let reportedModels = [];
|
|
25
|
-
const
|
|
26
|
-
? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
|
|
27
|
-
: undefined;
|
|
25
|
+
const steps = agent === 'opencode' || agent === 'kilo' ? createStepTelemetry() : undefined;
|
|
28
26
|
const update = (lines) => {
|
|
29
27
|
for (const line of lines) {
|
|
28
|
+
if (steps) {
|
|
29
|
+
try {
|
|
30
|
+
const event = JSON.parse(line);
|
|
31
|
+
if (event && typeof event === 'object' && !Array.isArray(event))
|
|
32
|
+
steps.consume(event);
|
|
33
|
+
}
|
|
34
|
+
catch { /* non-JSON diagnostics carry no usage */ }
|
|
35
|
+
}
|
|
30
36
|
if (pi) {
|
|
31
37
|
try {
|
|
32
38
|
const event = JSON.parse(line);
|
|
@@ -45,35 +51,6 @@ export function createTelemetryAccumulator(agent) {
|
|
|
45
51
|
// never add it to earlier results or to assistant-message snapshots.
|
|
46
52
|
if (next.tokens || next.partialUsage)
|
|
47
53
|
telemetry = next;
|
|
48
|
-
if (stepTotals && isStepFinish(line)) {
|
|
49
|
-
const usage = next.tokens ?? next.partialUsage;
|
|
50
|
-
if (usage) {
|
|
51
|
-
if (usage.inputTokens !== undefined) {
|
|
52
|
-
stepTotals.input += usage.inputTokens;
|
|
53
|
-
stepTotals.hasInput = true;
|
|
54
|
-
}
|
|
55
|
-
if (usage.outputTokens !== undefined) {
|
|
56
|
-
stepTotals.output += usage.outputTokens;
|
|
57
|
-
stepTotals.hasOutput = true;
|
|
58
|
-
}
|
|
59
|
-
if (usage.cachedInputTokens !== undefined) {
|
|
60
|
-
stepTotals.cached += usage.cachedInputTokens;
|
|
61
|
-
stepTotals.hasCached = true;
|
|
62
|
-
}
|
|
63
|
-
if (usage.cacheWriteInputTokens !== undefined) {
|
|
64
|
-
stepTotals.cacheWrite += usage.cacheWriteInputTokens;
|
|
65
|
-
stepTotals.hasCacheWrite = true;
|
|
66
|
-
}
|
|
67
|
-
if (usage.reasoningOutputTokens !== undefined) {
|
|
68
|
-
stepTotals.reasoning += usage.reasoningOutputTokens;
|
|
69
|
-
stepTotals.hasReasoning = true;
|
|
70
|
-
}
|
|
71
|
-
if (usage.totalCostUsd !== undefined) {
|
|
72
|
-
stepTotals.cost += usage.totalCostUsd;
|
|
73
|
-
stepTotals.hasCost = true;
|
|
74
|
-
}
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
54
|
}
|
|
78
55
|
};
|
|
79
56
|
return {
|
|
@@ -88,42 +65,13 @@ export function createTelemetryAccumulator(agent) {
|
|
|
88
65
|
trailing = '';
|
|
89
66
|
if (pi)
|
|
90
67
|
return pi.finish();
|
|
91
|
-
|
|
92
|
-
const latest = telemetry.tokens;
|
|
93
|
-
const inputTokens = stepTotals.hasInput ? stepTotals.input : latest?.inputTokens;
|
|
94
|
-
const outputTokens = stepTotals.hasOutput ? stepTotals.output : latest?.outputTokens;
|
|
95
|
-
const partialUsage = {
|
|
96
|
-
...(inputTokens !== undefined ? { inputTokens } : {}),
|
|
97
|
-
...(outputTokens !== undefined ? { outputTokens } : {}),
|
|
98
|
-
...(stepTotals.hasCached ? { cachedInputTokens: stepTotals.cached } : latest?.cachedInputTokens !== undefined ? { cachedInputTokens: latest.cachedInputTokens } : {}),
|
|
99
|
-
...(stepTotals.hasCacheWrite ? { cacheWriteInputTokens: stepTotals.cacheWrite } : latest?.cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens: latest.cacheWriteInputTokens } : {}),
|
|
100
|
-
...(stepTotals.hasReasoning ? { reasoningOutputTokens: stepTotals.reasoning } : latest?.reasoningOutputTokens !== undefined ? { reasoningOutputTokens: latest.reasoningOutputTokens } : {}),
|
|
101
|
-
...(stepTotals.hasCost ? { totalCostUsd: stepTotals.cost } : latest?.totalCostUsd !== undefined ? { totalCostUsd: latest.totalCostUsd } : {}),
|
|
102
|
-
...(latest?.model ? { model: latest.model } : {}),
|
|
103
|
-
};
|
|
104
|
-
if (inputTokens !== undefined && outputTokens !== undefined) {
|
|
105
|
-
telemetry = { usageAvailable: true, tokens: { ...partialUsage, inputTokens, outputTokens } };
|
|
106
|
-
}
|
|
107
|
-
else {
|
|
108
|
-
telemetry = { usageAvailable: false, partialUsage };
|
|
109
|
-
}
|
|
110
|
-
}
|
|
68
|
+
telemetry = steps?.finish() ?? telemetry;
|
|
111
69
|
if (telemetry.tokens) {
|
|
112
70
|
const { model: _model, ...tokens } = telemetry.tokens;
|
|
113
|
-
return {
|
|
71
|
+
return { ...telemetry, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
|
|
114
72
|
...(reportedModels.length > 1 ? { reportedModels } : {}) };
|
|
115
73
|
}
|
|
116
74
|
return { ...telemetry, ...(reportedModels.length ? { reportedModels } : {}) };
|
|
117
75
|
},
|
|
118
76
|
};
|
|
119
77
|
}
|
|
120
|
-
function isStepFinish(line) {
|
|
121
|
-
try {
|
|
122
|
-
const value = JSON.parse(line);
|
|
123
|
-
const part = value.part && typeof value.part === 'object' ? value.part : undefined;
|
|
124
|
-
return value.type === 'step_finish' || part?.type === 'step-finish';
|
|
125
|
-
}
|
|
126
|
-
catch {
|
|
127
|
-
return false;
|
|
128
|
-
}
|
|
129
|
-
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/** Runner defaults belong to the runner that selected them. Undefined overrides
|
|
2
|
+
* are absent; explicit false values (for example bare: false) still override. */
|
|
3
|
+
export function resolveProviderSelection(agent, defaults, overrides = {}) {
|
|
4
|
+
const inherited = defaults && (defaults.agent ?? 'codex') === agent ? defaults : {};
|
|
5
|
+
const result = {};
|
|
6
|
+
for (const key of ['provider', 'model', 'reasoningEffort', 'variant', 'bare', 'nativeMultiAgent']) {
|
|
7
|
+
const value = overrides[key] ?? inherited[key];
|
|
8
|
+
if (value !== undefined)
|
|
9
|
+
Object.assign(result, { [key]: value });
|
|
10
|
+
}
|
|
11
|
+
return result;
|
|
12
|
+
}
|
package/dist/agents/telemetry.js
CHANGED
|
@@ -13,6 +13,50 @@ function parseJson(value) {
|
|
|
13
13
|
function isRecord(value) {
|
|
14
14
|
return typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
15
15
|
}
|
|
16
|
+
/** OpenCode-family usage belongs to individual completed steps. Each field needs
|
|
17
|
+
* coverage of every step before it can be called a total. Shared by batch and
|
|
18
|
+
* streaming readers so an omitted measurement cannot turn into a measured zero. */
|
|
19
|
+
export function createStepTelemetry() {
|
|
20
|
+
let steps = 0;
|
|
21
|
+
const totals = {};
|
|
22
|
+
const counts = {};
|
|
23
|
+
return {
|
|
24
|
+
consume(event) {
|
|
25
|
+
const part = isRecord(event.part) ? event.part : undefined;
|
|
26
|
+
if (event.type !== 'step_finish' && part?.type !== 'step-finish')
|
|
27
|
+
return;
|
|
28
|
+
steps++;
|
|
29
|
+
const tokens = isRecord(part?.tokens) ? part.tokens : {};
|
|
30
|
+
const cache = isRecord(tokens.cache) ? tokens.cache : undefined;
|
|
31
|
+
const fields = {
|
|
32
|
+
inputTokens: tokens.input,
|
|
33
|
+
outputTokens: tokens.output,
|
|
34
|
+
cachedInputTokens: tokens.cached ?? tokens.cacheRead ?? cache?.read,
|
|
35
|
+
cacheWriteInputTokens: tokens.cacheWrite ?? cache?.write,
|
|
36
|
+
reasoningOutputTokens: tokens.reasoning,
|
|
37
|
+
totalCostUsd: part?.cost ?? event.cost,
|
|
38
|
+
};
|
|
39
|
+
for (const [field, raw] of Object.entries(fields)) {
|
|
40
|
+
const value = finite(raw);
|
|
41
|
+
if (value === undefined)
|
|
42
|
+
continue;
|
|
43
|
+
totals[field] = (totals[field] ?? 0) + value;
|
|
44
|
+
counts[field] = (counts[field] ?? 0) + 1;
|
|
45
|
+
}
|
|
46
|
+
},
|
|
47
|
+
finish() {
|
|
48
|
+
if (!steps)
|
|
49
|
+
return undefined;
|
|
50
|
+
const complete = counts.inputTokens === steps && counts.outputTokens === steps;
|
|
51
|
+
if (!complete)
|
|
52
|
+
return { usageAvailable: false, ...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}) };
|
|
53
|
+
const measured = Object.fromEntries(Object.entries(totals).filter(([field]) => counts[field] === steps));
|
|
54
|
+
const partialUsage = Object.fromEntries(Object.entries(totals).filter(([field]) => counts[field] !== steps));
|
|
55
|
+
return { usageAvailable: true, tokens: { ...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens },
|
|
56
|
+
...(Object.keys(partialUsage).length ? { partialUsage } : {}) };
|
|
57
|
+
},
|
|
58
|
+
};
|
|
59
|
+
}
|
|
16
60
|
function textContent(value) {
|
|
17
61
|
if (typeof value === 'string')
|
|
18
62
|
return value;
|
|
@@ -154,9 +198,7 @@ export function parseProviderTelemetry(agent, lines) {
|
|
|
154
198
|
let totalCostUsd;
|
|
155
199
|
let model;
|
|
156
200
|
let reportedModels = [];
|
|
157
|
-
const
|
|
158
|
-
? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
|
|
159
|
-
: undefined;
|
|
201
|
+
const steps = agent === 'opencode' || agent === 'kilo' ? createStepTelemetry() : undefined;
|
|
160
202
|
for (const line of lines) {
|
|
161
203
|
let parsed;
|
|
162
204
|
try {
|
|
@@ -168,45 +210,11 @@ export function parseProviderTelemetry(agent, lines) {
|
|
|
168
210
|
if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed))
|
|
169
211
|
continue;
|
|
170
212
|
const event = parsed;
|
|
213
|
+
steps?.consume(event);
|
|
171
214
|
if ((agent === 'qwen' || agent === 'claude') && event.parent_tool_use_id != null)
|
|
172
215
|
continue;
|
|
173
216
|
const message = event.message && typeof event.message === 'object' ? event.message : undefined;
|
|
174
217
|
const stats = event.stats && typeof event.stats === 'object' ? event.stats : undefined;
|
|
175
|
-
const part = event.part && typeof event.part === 'object' ? event.part : undefined;
|
|
176
|
-
const partTokens = part?.tokens && typeof part.tokens === 'object' ? part.tokens : undefined;
|
|
177
|
-
if (harnessTotals && (event.type === 'step_finish' || part?.type === 'step-finish') && partTokens) {
|
|
178
|
-
const cache = partTokens.cache && typeof partTokens.cache === 'object' ? partTokens.cache : undefined;
|
|
179
|
-
const stepInput = finite(partTokens.input);
|
|
180
|
-
const stepOutput = finite(partTokens.output);
|
|
181
|
-
const stepCached = finite(partTokens.cached ?? partTokens.cacheRead ?? cache?.read);
|
|
182
|
-
const stepCacheWrite = finite(partTokens.cacheWrite ?? cache?.write);
|
|
183
|
-
const stepReasoning = finite(partTokens.reasoning);
|
|
184
|
-
const stepCost = finite(part?.cost ?? event.cost);
|
|
185
|
-
if (stepInput !== undefined) {
|
|
186
|
-
harnessTotals.input += stepInput;
|
|
187
|
-
harnessTotals.hasInput = true;
|
|
188
|
-
}
|
|
189
|
-
if (stepOutput !== undefined) {
|
|
190
|
-
harnessTotals.output += stepOutput;
|
|
191
|
-
harnessTotals.hasOutput = true;
|
|
192
|
-
}
|
|
193
|
-
if (stepCached !== undefined) {
|
|
194
|
-
harnessTotals.cached += stepCached;
|
|
195
|
-
harnessTotals.hasCached = true;
|
|
196
|
-
}
|
|
197
|
-
if (stepCacheWrite !== undefined) {
|
|
198
|
-
harnessTotals.cacheWrite += stepCacheWrite;
|
|
199
|
-
harnessTotals.hasCacheWrite = true;
|
|
200
|
-
}
|
|
201
|
-
if (stepReasoning !== undefined) {
|
|
202
|
-
harnessTotals.reasoning += stepReasoning;
|
|
203
|
-
harnessTotals.hasReasoning = true;
|
|
204
|
-
}
|
|
205
|
-
if (stepCost !== undefined) {
|
|
206
|
-
harnessTotals.cost += stepCost;
|
|
207
|
-
harnessTotals.hasCost = true;
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
218
|
const usage = (event.usage && typeof event.usage === 'object'
|
|
211
219
|
? event.usage
|
|
212
220
|
: message?.usage && typeof message.usage === 'object'
|
|
@@ -277,20 +285,12 @@ export function parseProviderTelemetry(agent, lines) {
|
|
|
277
285
|
if (typeof eventModel === 'string' && eventModel && reportedModels.length <= 1)
|
|
278
286
|
model = eventModel;
|
|
279
287
|
}
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
cachedInputTokens = harnessTotals.cached;
|
|
287
|
-
if (harnessTotals.hasCacheWrite)
|
|
288
|
-
cacheWriteInputTokens = harnessTotals.cacheWrite;
|
|
289
|
-
if (harnessTotals.hasReasoning)
|
|
290
|
-
reasoningOutputTokens = harnessTotals.reasoning;
|
|
291
|
-
if (harnessTotals.hasCost)
|
|
292
|
-
totalCostUsd = harnessTotals.cost;
|
|
293
|
-
}
|
|
288
|
+
const stepUsage = steps?.finish();
|
|
289
|
+
if (stepUsage)
|
|
290
|
+
return { ...stepUsage,
|
|
291
|
+
...(stepUsage.tokens && model ? { tokens: { ...stepUsage.tokens, model } } : {}),
|
|
292
|
+
...(!stepUsage.tokens && model ? { reportedModels: [model] } : {}),
|
|
293
|
+
};
|
|
294
294
|
if (inputTokens === undefined || outputTokens === undefined) {
|
|
295
295
|
const partialUsage = {
|
|
296
296
|
...(inputTokens !== undefined ? { inputTokens } : {}),
|
package/dist/change/inbox.js
CHANGED
|
@@ -9,8 +9,9 @@ import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, w
|
|
|
9
9
|
import { basename, dirname, isAbsolute, join, resolve } from 'node:path';
|
|
10
10
|
import { z } from 'zod';
|
|
11
11
|
import { criterionCommandProblem, isAcceptanceCriterion, loadPrd, parsePrd, savePrd, validateDependencies } from '../loop/prd.js';
|
|
12
|
-
import { agentInvocation, buildWatchdogInvocation, isAgentAvailable,
|
|
12
|
+
import { agentInvocation, buildWatchdogInvocation, isAgentAvailable, runCapturedAgent, } from '../loop/runner.js';
|
|
13
13
|
import { commitPaths } from '../loop/git.js';
|
|
14
|
+
import { measureInvocation } from '../observability/invocation.js';
|
|
14
15
|
const ChangeRequestSchema = z.object({
|
|
15
16
|
version: z.literal(1),
|
|
16
17
|
id: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/),
|
|
@@ -184,7 +185,9 @@ export function runChangeApply(targetDir, opts) {
|
|
|
184
185
|
rmSync(proposal, { force: true });
|
|
185
186
|
const base = agentInvocation(opts.runner, buildChangePrompt(request, proposal, existing, brief), plannerDir, opts.permissions ?? 'safe', opts.selection);
|
|
186
187
|
const invocation = buildWatchdogInvocation(base, opts.timeoutMs ?? 0);
|
|
187
|
-
const result = (opts.
|
|
188
|
+
const result = measureInvocation({ root: targetDir, agent: opts.runner, role: 'change-planner', storyId: request.id, runId: `change:${request.id}`, selection: opts.selection, invocation,
|
|
189
|
+
execute: opts.run ?? (item => runCapturedAgent(opts.runner, item)),
|
|
190
|
+
});
|
|
188
191
|
if (!result.success)
|
|
189
192
|
return { ok: false, added: 0, summary: `planner failed: ${result.summary}`, changeId: request.id };
|
|
190
193
|
if (readFileSync(prdPath, 'utf8') !== existingText || (readPlanningFile(targetDir, '.yoke/plan.md', 80_000) ?? '') !== brief) {
|
|
@@ -246,7 +249,9 @@ export function runChangeApply(targetDir, opts) {
|
|
|
246
249
|
rmSync(reviewPath, { force: true });
|
|
247
250
|
const reviewBase = agentInvocation(reviewer, buildChangeReviewPrompt(request, reviewPath, appended), plannerDir, opts.permissions ?? 'safe', reviewer === opts.runner ? opts.selection : undefined);
|
|
248
251
|
const reviewInvocation = buildWatchdogInvocation(reviewBase, opts.timeoutMs ?? 0);
|
|
249
|
-
const reviewResult = (opts.
|
|
252
|
+
const reviewResult = measureInvocation({ root: targetDir, agent: reviewer, role: 'coverage-review', storyId: request.id, runId: `change:${request.id}`, selection: reviewer === opts.runner ? opts.selection : undefined, invocation: reviewInvocation,
|
|
253
|
+
execute: opts.review ?? (item => runCapturedAgent(reviewer, item)),
|
|
254
|
+
});
|
|
250
255
|
if (!reviewResult.success) {
|
|
251
256
|
return { ok: false, added: 0, summary: `coverage review failed: ${reviewResult.summary}`, changeId: request.id };
|
|
252
257
|
}
|