@hecer/yoke 1.21.1 → 1.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +48 -0
- package/README.md +8 -1
- package/TODOS.md +6 -0
- package/bench/analyze-codex-comparison.mjs +90 -17
- package/bench/compare-codex.mjs +159 -36
- package/bench/result-schema.mjs +132 -0
- package/canon/manifest.yaml +1 -1
- package/canon/skills/visual-verification/SKILL.md +25 -2
- package/canon/tools/codex-rtk-hook.mjs +6 -16
- package/dist/agents/pi-telemetry.js +2 -1
- package/dist/agents/process-streams.js +12 -64
- package/dist/agents/provider-selection.js +12 -0
- package/dist/agents/telemetry.js +52 -52
- package/dist/change/inbox.js +8 -3
- package/dist/check/command.js +69 -17
- package/dist/check/delivery.js +121 -0
- package/dist/cli.js +91 -3
- package/dist/code-intelligence/adapters/mcp.js +1 -0
- package/dist/code-intelligence/budgets.js +138 -0
- package/dist/code-intelligence/contracts.js +2 -0
- package/dist/code-intelligence/coordinator.js +159 -85
- package/dist/code-intelligence/evidence.js +87 -34
- package/dist/code-intelligence/index.js +1 -0
- package/dist/code-intelligence/mcp-client.js +25 -6
- package/dist/code-intelligence/mcp-server.js +14 -11
- package/dist/code-intelligence/preflight.js +71 -0
- package/dist/dashboard/analytics.js +5 -3
- package/dist/goals/command.js +183 -53
- package/dist/goals/usage.js +87 -0
- package/dist/loop/cache-isolation.js +36 -0
- package/dist/loop/candidate-cleanup.js +47 -17
- package/dist/loop/candidates.js +17 -11
- package/dist/loop/dispatcher.js +89 -26
- package/dist/loop/failure.js +104 -0
- package/dist/loop/gate-snapshot.js +19 -0
- package/dist/loop/git.js +1 -1
- package/dist/loop/loop.js +124 -70
- package/dist/loop/parallel-adapters.js +57 -6
- package/dist/loop/parallel-command.js +49 -7
- package/dist/loop/proof-retention.js +70 -0
- package/dist/loop/recovery.js +23 -5
- package/dist/loop/reporter.js +22 -5
- package/dist/loop/run-command.js +101 -47
- package/dist/loop/runner.js +6 -5
- package/dist/loop/worker.js +152 -91
- package/dist/observability/history.js +2 -1
- package/dist/observability/invocation.js +42 -0
- package/dist/observability/local-report.js +120 -0
- package/dist/observability/usage.js +19 -0
- package/dist/prd/command.js +20 -7
- package/dist/prd/decompose.js +5 -2
- package/dist/retrofit/config.js +29 -2
- package/dist/retrofit/gitignore.js +12 -0
- package/dist/retrofit/planners/codex.js +20 -20
- package/dist/routing/attempts.js +241 -0
- package/dist/routing/capability.js +13 -9
- package/dist/routing/optimization.js +73 -0
- package/dist/routing/registry.js +7 -1
- package/dist/routing/router.js +282 -127
- package/dist/setup/command.js +8 -2
- package/dist/smoke/command.js +387 -85
- package/dist/update/check.js +1 -1
- package/docs/BENCHMARK-MANIFEST.md +131 -0
- package/docs/CODE-INTELLIGENCE.md +43 -1
- package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
- package/docs/DELIVERY-JOURNEYS.md +206 -0
- package/docs/ECONOMIC-ROUTING.md +180 -0
- package/docs/GOALS.md +61 -4
- package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
- package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
- package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
- package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
- package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
- package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
- package/docs/parallel-execution.md +37 -9
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
- package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
package/dist/smoke/command.js
CHANGED
|
@@ -1,7 +1,10 @@
|
|
|
1
|
-
import { mkdirSync, rmSync, renameSync } from 'node:fs';
|
|
1
|
+
import { lstatSync, mkdirSync, rmSync, renameSync, writeFileSync } from 'node:fs';
|
|
2
2
|
import { join, resolve } from 'node:path';
|
|
3
3
|
import { createRequire } from 'node:module';
|
|
4
|
+
import { createHash } from 'node:crypto';
|
|
4
5
|
import { loadConfig } from '../retrofit/config.js';
|
|
6
|
+
import { workspaceFingerprint } from '../workspace/fingerprint.js';
|
|
7
|
+
import { statePath } from '../workspace/state.js';
|
|
5
8
|
const CONFIG_GUIDANCE = [
|
|
6
9
|
'No smoke flows configured. Add a smoke section to .yoke/config.yaml, e.g.:',
|
|
7
10
|
'',
|
|
@@ -13,26 +16,41 @@ const CONFIG_GUIDANCE = [
|
|
|
13
16
|
' landmark: "main h1"',
|
|
14
17
|
].join('\n');
|
|
15
18
|
export async function launchPlaywright(targetDir) {
|
|
19
|
+
const req = createRequire(join(resolve(targetDir), 'package.json'));
|
|
16
20
|
try {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
// short paths (e.g. RUNNER~1 on CI) and test-runner import interception.
|
|
22
|
-
const req = createRequire(join(resolve(targetDir), 'package.json'));
|
|
23
|
-
const pw = req('playwright');
|
|
24
|
-
const chromium = pw.chromium ?? pw.default?.chromium;
|
|
25
|
-
if (!chromium)
|
|
21
|
+
req.resolve('playwright');
|
|
22
|
+
}
|
|
23
|
+
catch (error) {
|
|
24
|
+
if (error.code === 'MODULE_NOT_FOUND')
|
|
26
25
|
return null;
|
|
27
|
-
|
|
26
|
+
throw Object.assign(new Error(`Playwright resolution failed: ${String(error)}`, { cause: error }), { code: 'browser-package-failed' });
|
|
28
27
|
}
|
|
29
|
-
|
|
30
|
-
|
|
28
|
+
// Native CJS loading also preserves Windows short-path compatibility.
|
|
29
|
+
let chromium;
|
|
30
|
+
try {
|
|
31
|
+
const pw = req('playwright');
|
|
32
|
+
chromium = pw.chromium ?? pw.default?.chromium;
|
|
33
|
+
}
|
|
34
|
+
catch (error) {
|
|
35
|
+
throw Object.assign(new Error(`Playwright package could not load: ${String(error)}`, { cause: error }), { code: 'browser-package-failed' });
|
|
36
|
+
}
|
|
37
|
+
if (!chromium)
|
|
38
|
+
throw Object.assign(new Error('Installed Playwright package has no chromium export'), { code: 'browser-export-missing' });
|
|
39
|
+
try {
|
|
40
|
+
return await chromium.launch({ headless: true, timeout: 30_000 });
|
|
41
|
+
}
|
|
42
|
+
catch (error) {
|
|
43
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
44
|
+
const code = /executable doesn't exist|executable does not exist/i.test(detail) ? 'browser-binary-missing'
|
|
45
|
+
: /no usable sandbox|running as root without --no-sandbox/i.test(detail) ? 'browser-sandbox-failed'
|
|
46
|
+
: /\bEPERM\b|\bEACCES\b|permission denied|operation not permitted/i.test(detail) ? 'browser-permission-denied'
|
|
47
|
+
: 'browser-launch-failed';
|
|
48
|
+
throw Object.assign(new Error(detail, { cause: error }), { code });
|
|
31
49
|
}
|
|
32
50
|
}
|
|
33
51
|
// Flow names come from user config and become filenames — keep them safe.
|
|
34
52
|
function safeName(name) {
|
|
35
|
-
return name.replace(/[^\w.-]+/g, '-');
|
|
53
|
+
return name.replace(/[^\w.-]+/g, '-').slice(0, 120) || 'flow';
|
|
36
54
|
}
|
|
37
55
|
// The label names a directory that gets rmSync'd recursively — it must never
|
|
38
56
|
// carry path semantics ('..', separators). Dots are stripped entirely so a
|
|
@@ -41,6 +59,230 @@ export function safeLabel(label) {
|
|
|
41
59
|
const cleaned = label.replace(/[^\w-]+/g, '-').replace(/^-+|-+$/g, '');
|
|
42
60
|
return cleaned || 'latest';
|
|
43
61
|
}
|
|
62
|
+
async function verifyServedSource(baseUrl, identity) {
|
|
63
|
+
const origin = new URL(baseUrl);
|
|
64
|
+
const url = new URL(identity.path, origin);
|
|
65
|
+
if (!['http:', 'https:'].includes(url.protocol) || url.origin !== origin.origin)
|
|
66
|
+
return false;
|
|
67
|
+
const response = await fetch(url, { signal: AbortSignal.timeout(5000), redirect: 'error' });
|
|
68
|
+
if (!response.ok || !response.body)
|
|
69
|
+
return false;
|
|
70
|
+
const reader = response.body.getReader(), hash = createHash('sha256');
|
|
71
|
+
let bytes = 0;
|
|
72
|
+
try {
|
|
73
|
+
while (true) {
|
|
74
|
+
const { done, value } = await reader.read();
|
|
75
|
+
if (done)
|
|
76
|
+
break;
|
|
77
|
+
bytes += value.byteLength;
|
|
78
|
+
if (bytes > 1024 * 1024)
|
|
79
|
+
return false;
|
|
80
|
+
hash.update(value);
|
|
81
|
+
}
|
|
82
|
+
return hash.digest('hex') === identity.sha256;
|
|
83
|
+
}
|
|
84
|
+
finally {
|
|
85
|
+
await reader.cancel().catch(() => { });
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
class SmokeFailure extends Error {
|
|
89
|
+
code;
|
|
90
|
+
constructor(code) {
|
|
91
|
+
super(code);
|
|
92
|
+
this.code = code;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
function failureCode(error, fallback) {
|
|
96
|
+
if (error instanceof SmokeFailure)
|
|
97
|
+
return error.code;
|
|
98
|
+
return error?.name === 'TimeoutError' ? 'timeout' : fallback;
|
|
99
|
+
}
|
|
100
|
+
async function bounded(operation, timeout) {
|
|
101
|
+
if (timeout <= 0)
|
|
102
|
+
throw new SmokeFailure('timeout');
|
|
103
|
+
const deadline = Date.now() + timeout;
|
|
104
|
+
let timer;
|
|
105
|
+
try {
|
|
106
|
+
const result = await Promise.race([
|
|
107
|
+
Promise.resolve().then(operation),
|
|
108
|
+
new Promise((_, reject) => { timer = setTimeout(() => reject(new SmokeFailure('timeout')), timeout); }),
|
|
109
|
+
]);
|
|
110
|
+
if (Date.now() >= deadline)
|
|
111
|
+
throw new SmokeFailure('timeout');
|
|
112
|
+
return result;
|
|
113
|
+
}
|
|
114
|
+
finally {
|
|
115
|
+
if (timer)
|
|
116
|
+
clearTimeout(timer);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
async function measured(stage, operation, fallback) {
|
|
120
|
+
const started = Date.now();
|
|
121
|
+
try {
|
|
122
|
+
await operation();
|
|
123
|
+
stage.status = 'passed';
|
|
124
|
+
}
|
|
125
|
+
catch (error) {
|
|
126
|
+
stage.status = 'failed';
|
|
127
|
+
stage.failure = failureCode(error, fallback);
|
|
128
|
+
throw new SmokeFailure(stage.failure);
|
|
129
|
+
}
|
|
130
|
+
finally {
|
|
131
|
+
stage.durationMs = Date.now() - started;
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
async function executeStep(page, step, baseUrl, timeout, fillValue) {
|
|
135
|
+
switch (step.action) {
|
|
136
|
+
case 'click':
|
|
137
|
+
if (!page.click)
|
|
138
|
+
throw new SmokeFailure('unsupported-action');
|
|
139
|
+
await page.click(step.selector, { timeout });
|
|
140
|
+
return;
|
|
141
|
+
case 'fill':
|
|
142
|
+
if (fillValue === undefined || fillValue.length > 8192)
|
|
143
|
+
throw new SmokeFailure('fill-value-unavailable');
|
|
144
|
+
if (!page.fill)
|
|
145
|
+
throw new SmokeFailure('unsupported-action');
|
|
146
|
+
await page.fill(step.selector, fillValue, { timeout });
|
|
147
|
+
return;
|
|
148
|
+
case 'press':
|
|
149
|
+
if (!page.press)
|
|
150
|
+
throw new SmokeFailure('unsupported-action');
|
|
151
|
+
await page.press(step.selector, step.key, { timeout });
|
|
152
|
+
return;
|
|
153
|
+
case 'expect-visible':
|
|
154
|
+
await page.waitForSelector(step.selector, { state: 'visible', timeout });
|
|
155
|
+
return;
|
|
156
|
+
case 'expect-text': {
|
|
157
|
+
if (!page.textContent)
|
|
158
|
+
throw new SmokeFailure('unsupported-action');
|
|
159
|
+
const deadline = Date.now() + timeout;
|
|
160
|
+
await page.waitForSelector(step.selector, { state: 'visible', timeout });
|
|
161
|
+
const normalize = (value) => value.replace(/\s+/gu, ' ').trim();
|
|
162
|
+
const expected = normalize(step.text);
|
|
163
|
+
// Both a fixed attempt bound and the enclosing deadline apply. No unbounded
|
|
164
|
+
// polling, even when a test adapter ignores Playwright's own timeout option.
|
|
165
|
+
for (let attempt = 0; attempt <= Math.ceil(timeout / 100); attempt++) {
|
|
166
|
+
const remaining = deadline - Date.now();
|
|
167
|
+
if (remaining <= 0)
|
|
168
|
+
break;
|
|
169
|
+
const actual = normalize(await page.textContent(step.selector, { timeout: remaining }) ?? '');
|
|
170
|
+
if (Date.now() < deadline && (step.exact ? actual === expected : actual.includes(expected)))
|
|
171
|
+
return;
|
|
172
|
+
await new Promise(resolveValue => setTimeout(resolveValue, Math.max(0, Math.min(100, deadline - Date.now()))));
|
|
173
|
+
}
|
|
174
|
+
throw new SmokeFailure('timeout');
|
|
175
|
+
}
|
|
176
|
+
case 'expect-url': {
|
|
177
|
+
if (!page.waitForURL)
|
|
178
|
+
throw new SmokeFailure('unsupported-action');
|
|
179
|
+
const expected = new URL(step.url, baseUrl).href;
|
|
180
|
+
await page.waitForURL(url => url.href === expected, { waitUntil: 'load', timeout });
|
|
181
|
+
return;
|
|
182
|
+
}
|
|
183
|
+
case 'reload': {
|
|
184
|
+
if (!page.reload)
|
|
185
|
+
throw new SmokeFailure('unsupported-action');
|
|
186
|
+
const response = await page.reload({ waitUntil: 'load', timeout });
|
|
187
|
+
if (response && !response.ok())
|
|
188
|
+
throw new SmokeFailure('http-error');
|
|
189
|
+
return;
|
|
190
|
+
}
|
|
191
|
+
default: {
|
|
192
|
+
const unexpected = step;
|
|
193
|
+
void unexpected;
|
|
194
|
+
throw new SmokeFailure('unsupported-action');
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
async function runFlow(browser, flow, baseUrl, proofDir, fileStem, fillValues, name) {
|
|
199
|
+
const started = Date.now(), deadline = started + (flow.timeoutMs ?? 60_000);
|
|
200
|
+
const report = {
|
|
201
|
+
name, status: 'failed', durationMs: 0, navigation: { status: 'skipped', durationMs: 0 },
|
|
202
|
+
...(flow.landmark ? { landmark: { status: 'skipped', durationMs: 0 } } : {}),
|
|
203
|
+
steps: (flow.steps ?? []).map((step, index) => ({ index: index + 1, action: step.action, status: 'skipped', durationMs: 0 })),
|
|
204
|
+
};
|
|
205
|
+
let context, page, consoleErrors = 0;
|
|
206
|
+
let video = null;
|
|
207
|
+
const remaining = (limit) => Math.min(limit, deadline - Date.now());
|
|
208
|
+
const assertNoConsoleErrors = () => { if (consoleErrors > 0)
|
|
209
|
+
throw new SmokeFailure('console-errors'); };
|
|
210
|
+
try {
|
|
211
|
+
context = await bounded(() => browser.newContext({ recordVideo: { dir: join(proofDir, '.video-tmp') }, viewport: { width: 1280, height: 720 } }), remaining(30_000));
|
|
212
|
+
page = await bounded(() => context.newPage(), remaining(30_000));
|
|
213
|
+
const activePage = page;
|
|
214
|
+
// Browser diagnostics can echo input values. Count them; never persist their
|
|
215
|
+
// text, call logs, selectors or parameters in the machine-readable report.
|
|
216
|
+
page.on('console', msg => { if (msg?.type?.() === 'error')
|
|
217
|
+
consoleErrors++; });
|
|
218
|
+
page.on('pageerror', () => { consoleErrors++; });
|
|
219
|
+
await measured(report.navigation, async () => {
|
|
220
|
+
const timeout = remaining(30_000);
|
|
221
|
+
const response = await bounded(() => activePage.goto(baseUrl + flow.path, { waitUntil: 'load', timeout }), timeout);
|
|
222
|
+
if (response && !response.ok())
|
|
223
|
+
throw new SmokeFailure('http-error');
|
|
224
|
+
assertNoConsoleErrors();
|
|
225
|
+
}, 'navigation-failed');
|
|
226
|
+
if (flow.landmark && report.landmark) {
|
|
227
|
+
await measured(report.landmark, async () => {
|
|
228
|
+
const timeout = remaining(10_000);
|
|
229
|
+
await bounded(() => activePage.waitForSelector(flow.landmark, { state: 'visible', timeout }), timeout);
|
|
230
|
+
assertNoConsoleErrors();
|
|
231
|
+
}, 'landmark-missing');
|
|
232
|
+
}
|
|
233
|
+
for (const [index, step] of (flow.steps ?? []).entries()) {
|
|
234
|
+
await measured(report.steps[index], async () => {
|
|
235
|
+
const timeout = remaining(step.timeoutMs ?? 10_000);
|
|
236
|
+
await bounded(() => executeStep(activePage, step, baseUrl, timeout, fillValues.get(step)), timeout);
|
|
237
|
+
assertNoConsoleErrors();
|
|
238
|
+
}, 'step-failed');
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
catch (error) {
|
|
242
|
+
report.failure = failureCode(error, 'context-failed');
|
|
243
|
+
}
|
|
244
|
+
finally {
|
|
245
|
+
if (page) {
|
|
246
|
+
try {
|
|
247
|
+
await bounded(() => page.screenshot({ path: join(proofDir, `${fileStem}.png`), fullPage: true, timeout: 5000 }), 5000);
|
|
248
|
+
const saved = lstatSync(join(proofDir, `${fileStem}.png`));
|
|
249
|
+
if (!saved.isFile() || saved.size === 0)
|
|
250
|
+
throw new SmokeFailure('screenshot-failed');
|
|
251
|
+
report.screenshot = `${fileStem}.png`;
|
|
252
|
+
}
|
|
253
|
+
catch {
|
|
254
|
+
report.failure ??= 'screenshot-failed';
|
|
255
|
+
}
|
|
256
|
+
try {
|
|
257
|
+
video = page.video();
|
|
258
|
+
}
|
|
259
|
+
catch { /* Video is best-effort evidence. */ }
|
|
260
|
+
}
|
|
261
|
+
if (context) {
|
|
262
|
+
try {
|
|
263
|
+
await bounded(() => context.close(), 5000);
|
|
264
|
+
}
|
|
265
|
+
catch {
|
|
266
|
+
report.failure ??= 'context-cleanup-failed';
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
if (video) {
|
|
270
|
+
try {
|
|
271
|
+
const path = await bounded(() => video.path(), 5000);
|
|
272
|
+
if (report.failure) {
|
|
273
|
+
renameSync(path, join(proofDir, `${fileStem}.webm`));
|
|
274
|
+
report.failureVideo = `${fileStem}.webm`;
|
|
275
|
+
}
|
|
276
|
+
else
|
|
277
|
+
rmSync(path, { force: true });
|
|
278
|
+
}
|
|
279
|
+
catch { /* Preserve the original failure if video is unavailable. */ }
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
report.status = report.failure ? 'failed' : 'passed';
|
|
283
|
+
report.durationMs = Date.now() - started;
|
|
284
|
+
return report;
|
|
285
|
+
}
|
|
44
286
|
export async function runFlowSmoke(targetDir, opts = {}) {
|
|
45
287
|
const config = loadConfig(targetDir);
|
|
46
288
|
const smoke = config?.smoke;
|
|
@@ -51,91 +293,151 @@ export async function runFlowSmoke(targetDir, opts = {}) {
|
|
|
51
293
|
const baseUrl = opts.url ?? smoke.baseUrl;
|
|
52
294
|
const label = safeLabel(opts.label ?? process.env.YOKE_STORY ?? 'latest');
|
|
53
295
|
const proofRel = join('.yoke', 'proof', label);
|
|
54
|
-
|
|
296
|
+
let proofDir, sourceFingerprint;
|
|
297
|
+
try {
|
|
298
|
+
proofDir = statePath(targetDir, 'proof', label);
|
|
299
|
+
sourceFingerprint = workspaceFingerprint(targetDir);
|
|
300
|
+
}
|
|
301
|
+
catch {
|
|
302
|
+
console.error('Smoke evidence path or source fingerprint could not be verified; previous evidence was preserved.');
|
|
303
|
+
return 2;
|
|
304
|
+
}
|
|
305
|
+
const started = Date.now();
|
|
306
|
+
const fillValues = new Map();
|
|
307
|
+
for (const flow of smoke.flows)
|
|
308
|
+
for (const step of flow.steps ?? []) {
|
|
309
|
+
if (step.action === 'fill')
|
|
310
|
+
fillValues.set(step, step.value ?? (step.valueEnv ? process.env[step.valueEnv] : undefined));
|
|
311
|
+
}
|
|
312
|
+
const configDigest = createHash('sha256').update(JSON.stringify({ smoke, baseUrl, resolvedFillValues: [...fillValues.values()] })).digest('hex');
|
|
313
|
+
const redact = (text) => [...fillValues.values()].filter((value) => !!value).sort((a, b) => b.length - a.length).reduce((current, value) => current.split(value).join('[redacted]'), text);
|
|
55
314
|
const launch = opts.launch ?? launchPlaywright;
|
|
56
|
-
|
|
315
|
+
if (!opts.launch && !smoke.sourceIdentity) {
|
|
316
|
+
console.error('Smoke sourceIdentity is required for a production browser run: configure a served path and expected SHA-256 for this source build; previous evidence was preserved.');
|
|
317
|
+
return 2;
|
|
318
|
+
}
|
|
319
|
+
if (smoke.sourceIdentity) {
|
|
320
|
+
let matched = false;
|
|
321
|
+
try {
|
|
322
|
+
matched = await verifyServedSource(baseUrl, smoke.sourceIdentity);
|
|
323
|
+
}
|
|
324
|
+
catch { /* Unavailable identity cannot pass. */ }
|
|
325
|
+
if (!matched) {
|
|
326
|
+
console.error('Smoke served-source identity could not be verified; previous evidence was preserved.');
|
|
327
|
+
return 2;
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
let browser;
|
|
331
|
+
try {
|
|
332
|
+
browser = await bounded(() => launch(targetDir), 30_000);
|
|
333
|
+
}
|
|
334
|
+
catch (error) {
|
|
335
|
+
const rawCode = error?.code;
|
|
336
|
+
const code = typeof rawCode === 'string' && ['browser-package-failed', 'browser-export-missing', 'browser-binary-missing', 'browser-sandbox-failed', 'browser-permission-denied'].includes(rawCode) ? rawCode : 'browser-launch-failed';
|
|
337
|
+
const detail = redact(error instanceof Error ? error.message : String(error))
|
|
338
|
+
.replace(/(https?:\/\/)[^\s/@]+:[^\s/@]+@/gi, '$1[redacted]@')
|
|
339
|
+
.replace(/\b((?:access_|refresh_|id_)?token|password|secret|api[_-]?key)=([^\s&]+)/gi, '$1=[redacted]')
|
|
340
|
+
.replace(/\b(authorization\s*:\s*(?:bearer|basic)\s+)[^\s,;]+/gi, '$1[redacted]');
|
|
341
|
+
console.error(`Smoke browser could not start (${code}): ${detail.slice(0, 2000)}; previous evidence was preserved.`);
|
|
342
|
+
return 2;
|
|
343
|
+
}
|
|
57
344
|
if (!browser) {
|
|
58
345
|
console.error(`Playwright not found in ${targetDir}. Install it: npm i -D playwright && npx playwright install chromium`);
|
|
59
346
|
return 2;
|
|
60
347
|
}
|
|
61
348
|
// Wipe only once the run is actually going to happen — an exit-2 run must
|
|
62
349
|
// not destroy the previous run's evidence.
|
|
63
|
-
|
|
64
|
-
|
|
350
|
+
try {
|
|
351
|
+
statePath(targetDir, 'proof', label);
|
|
352
|
+
rmSync(proofDir, { recursive: true, force: true }); // fresh evidence per run
|
|
353
|
+
mkdirSync(proofDir, { recursive: true });
|
|
354
|
+
}
|
|
355
|
+
catch {
|
|
356
|
+
try {
|
|
357
|
+
await bounded(() => browser.close(), 5000);
|
|
358
|
+
}
|
|
359
|
+
catch { /* Report setup failed. */ }
|
|
360
|
+
console.error('Smoke evidence directory could not be prepared.');
|
|
361
|
+
return 2;
|
|
362
|
+
}
|
|
65
363
|
const videoTmp = join(proofDir, '.video-tmp');
|
|
66
|
-
let
|
|
364
|
+
let baseOrigin = 'unavailable', browserVersion;
|
|
67
365
|
try {
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
reason = e.message.split('\n')[0];
|
|
99
|
-
}
|
|
100
|
-
// The screenshot IS the evidence — taken on success AND failure; a crashed
|
|
101
|
-
// page must not mask the original failure.
|
|
102
|
-
const shotName = `${safeName(flow.name)}.png`;
|
|
103
|
-
let shotOk = false;
|
|
104
|
-
try {
|
|
105
|
-
await page.screenshot({ path: join(proofDir, shotName), fullPage: true });
|
|
106
|
-
shotOk = true;
|
|
107
|
-
}
|
|
108
|
-
catch { /* keep the original reason */ }
|
|
109
|
-
const video = page.video();
|
|
110
|
-
await context.close();
|
|
111
|
-
let videoKept = false;
|
|
112
|
-
if (video) {
|
|
113
|
-
try {
|
|
114
|
-
const vpath = await video.path();
|
|
115
|
-
if (reason) {
|
|
116
|
-
renameSync(vpath, join(proofDir, `${safeName(flow.name)}.webm`));
|
|
117
|
-
videoKept = true;
|
|
118
|
-
}
|
|
119
|
-
else {
|
|
120
|
-
rmSync(vpath, { force: true });
|
|
121
|
-
}
|
|
122
|
-
}
|
|
123
|
-
catch { /* video is best-effort evidence */ }
|
|
124
|
-
}
|
|
125
|
-
if (reason) {
|
|
126
|
-
const saved = [shotOk ? 'screenshot' : null, videoKept ? 'video' : null].filter(Boolean).join(' + ');
|
|
127
|
-
console.log(`✘ ${flow.name} — ${reason}${saved ? ` (${saved} saved under ${proofRel})` : ''}`);
|
|
128
|
-
}
|
|
129
|
-
else {
|
|
130
|
-
green++;
|
|
131
|
-
console.log(`✔ ${flow.name} (screenshot: ${join(proofRel, shotName)})`);
|
|
366
|
+
baseOrigin = new URL(baseUrl).origin;
|
|
367
|
+
}
|
|
368
|
+
catch { /* Navigation reports the invalid address. */ }
|
|
369
|
+
try {
|
|
370
|
+
const version = browser.version?.();
|
|
371
|
+
if (version)
|
|
372
|
+
browserVersion = redact(version.slice(0, 120));
|
|
373
|
+
}
|
|
374
|
+
catch { /* Optional runtime metadata. */ }
|
|
375
|
+
const report = {
|
|
376
|
+
serverIdentity: smoke.sourceIdentity ? 'verified' : 'unverified',
|
|
377
|
+
version: 1, startedAt: new Date(started).toISOString(), generatedAt: '', durationMs: 0, status: 'failed',
|
|
378
|
+
sourceFingerprint, afterSourceFingerprint: null, sourceStable: false, configDigest,
|
|
379
|
+
environment: { platform: process.platform, architecture: process.arch, nodeVersion: process.version, driver: opts.launch ? 'injected' : 'playwright-chromium', ...(browserVersion ? { browserVersion } : {}), baseOrigin: redact(baseOrigin), viewport: { width: 1280, height: 720 } },
|
|
380
|
+
flows: [],
|
|
381
|
+
};
|
|
382
|
+
const filenames = new Set();
|
|
383
|
+
try {
|
|
384
|
+
for (const [index, flow] of smoke.flows.entries()) {
|
|
385
|
+
const baseName = safeName(redact(flow.name));
|
|
386
|
+
let fileStem = baseName, suffix = index + 1;
|
|
387
|
+
while (filenames.has(fileStem.toLowerCase()))
|
|
388
|
+
fileStem = `${baseName}-${suffix++}`;
|
|
389
|
+
filenames.add(fileStem.toLowerCase());
|
|
390
|
+
const result = await runFlow(browser, flow, baseUrl, proofDir, fileStem, fillValues, redact(flow.name));
|
|
391
|
+
report.flows.push(result);
|
|
392
|
+
if (result.status === 'failed') {
|
|
393
|
+
const saved = [result.screenshot ? 'screenshot' : null, result.failureVideo ? 'video' : null].filter(Boolean).join(' + ');
|
|
394
|
+
const failedStep = result.steps.find(step => step.status === 'failed');
|
|
395
|
+
console.log(`✘ ${result.name} — ${failedStep ? `step ${failedStep.index} (${failedStep.action}): ` : ''}${result.failure}${saved ? ` (${saved} saved under ${proofRel})` : ''}`);
|
|
132
396
|
}
|
|
397
|
+
else
|
|
398
|
+
console.log(`✔ ${result.name} (screenshot: ${join(proofRel, result.screenshot)})`);
|
|
133
399
|
}
|
|
134
400
|
}
|
|
135
401
|
finally {
|
|
136
|
-
|
|
402
|
+
try {
|
|
403
|
+
await bounded(() => browser.close(), 5000);
|
|
404
|
+
}
|
|
405
|
+
catch {
|
|
406
|
+
report.failure = 'browser-cleanup-failed';
|
|
407
|
+
}
|
|
137
408
|
rmSync(videoTmp, { recursive: true, force: true });
|
|
138
409
|
}
|
|
410
|
+
try {
|
|
411
|
+
report.afterSourceFingerprint = workspaceFingerprint(targetDir);
|
|
412
|
+
}
|
|
413
|
+
catch { /* Missing identity prevents a pass. */ }
|
|
414
|
+
if (smoke.sourceIdentity) {
|
|
415
|
+
let matched = false;
|
|
416
|
+
try {
|
|
417
|
+
matched = await verifyServedSource(baseUrl, smoke.sourceIdentity);
|
|
418
|
+
}
|
|
419
|
+
catch { /* Fail closed. */ }
|
|
420
|
+
if (!matched) {
|
|
421
|
+
report.serverIdentity = 'failed';
|
|
422
|
+
report.failure ??= 'server-identity-failed';
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
report.sourceStable = report.sourceFingerprint === report.afterSourceFingerprint;
|
|
426
|
+
if (!report.sourceStable)
|
|
427
|
+
report.failure ??= 'source-changed-or-unavailable';
|
|
428
|
+
const green = report.flows.filter(flow => flow.status === 'passed').length;
|
|
429
|
+
report.status = !report.failure && green === smoke.flows.length ? 'passed' : 'failed';
|
|
430
|
+
report.generatedAt = new Date().toISOString();
|
|
431
|
+
report.durationMs = Date.now() - started;
|
|
432
|
+
try {
|
|
433
|
+
writeFileSync(statePath(targetDir, 'proof', label, 'report.json'), JSON.stringify(report, null, 2), { mode: 0o600 });
|
|
434
|
+
}
|
|
435
|
+
catch {
|
|
436
|
+
console.error('Smoke report could not be saved.');
|
|
437
|
+
return 2;
|
|
438
|
+
}
|
|
439
|
+
if (report.failure)
|
|
440
|
+
console.log(`✘ Flow-smoke evidence: ${report.failure}`);
|
|
139
441
|
console.log(`Flow-smoke: ${green}/${smoke.flows.length} flows green — proof: ${proofRel}`);
|
|
140
|
-
return
|
|
442
|
+
return report.status === 'passed' ? 0 : 1;
|
|
141
443
|
}
|
package/dist/update/check.js
CHANGED
|
@@ -69,7 +69,7 @@ export function maybeNotifyUpdate(currentVersion, opts = {}) {
|
|
|
69
69
|
if (env.YOKE_NO_UPDATE_CHECK || env.CI)
|
|
70
70
|
return;
|
|
71
71
|
const argv = opts.argv ?? process.argv;
|
|
72
|
-
if (argv.
|
|
72
|
+
if (argv.some(arg => ['--json', '--help', '-h'].includes(arg)))
|
|
73
73
|
return;
|
|
74
74
|
const tty = opts.tty ?? process.stderr.isTTY === true;
|
|
75
75
|
if (!tty)
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# Benchmark comparison manifests
|
|
2
|
+
|
|
3
|
+
Yoke 1.22.0 adds a versioned provenance and comparability contract to the direct
|
|
4
|
+
Codex comparison tools. This changes the tooling, not the historical results.
|
|
5
|
+
No authenticated development run or new savings measurement is implied.
|
|
6
|
+
|
|
7
|
+
## What a new result records
|
|
8
|
+
|
|
9
|
+
The result document has schemaVersion 2 and a manifest with schemaVersion 1.
|
|
10
|
+
The manifest declares a unique comparison identity, its purpose (workflow or
|
|
11
|
+
controlled), the expected arms, the planned paired repeat count, and every
|
|
12
|
+
condition that is intentionally allowed to differ between arms. Each declared
|
|
13
|
+
difference needs a field name and a reason.
|
|
14
|
+
|
|
15
|
+
Each run contains its own context:
|
|
16
|
+
|
|
17
|
+
| Section | Recorded conditions |
|
|
18
|
+
| --- | --- |
|
|
19
|
+
| Source | Package version, Git commit, dirty state, SHA-256 build digest and whether source/build inputs stayed stable during the run |
|
|
20
|
+
| Fixture | Fixture identity and SHA-256 digests of the seed, original acceptance files and shared requirements |
|
|
21
|
+
| Execution | Provider, requested model, provider-reported model identities, effort, routing, native delegation, native goals, worker count and workflow |
|
|
22
|
+
| Prompt | Digest and scope of the submitted input: direct provider prompt or Yoke PRD/configuration bundle |
|
|
23
|
+
| Startup | Permissions, bare mode, ignore-rules, commit policy, isolation, timeout policy and user-state policy |
|
|
24
|
+
| Environment | Platform, architecture, Node version, provider CLI version, hashed host identity, background-load policy and whether skills/plugins were audited |
|
|
25
|
+
|
|
26
|
+
The build digest hashes the actual runtime files, canon, adapters/hooks, package
|
|
27
|
+
metadata, dependency lockfile and comparison tooling. It is recorded even for a
|
|
28
|
+
clean checkout, because an ignored dist directory can be stale. Dirty work is
|
|
29
|
+
therefore identified by both its base commit and runtime digest. Changes during a
|
|
30
|
+
run invalidate the stable-source condition. Missing Git or CLI provenance remains
|
|
31
|
+
unknown; it is never filled from a version written into an older report.
|
|
32
|
+
|
|
33
|
+
These digests identify recorded inputs. They are not signatures and do not prove
|
|
34
|
+
that a provider served a particular model. Reported model identity comes from
|
|
35
|
+
provider telemetry, never from the requested alias as a fallback.
|
|
36
|
+
|
|
37
|
+
## Submitted inputs and effective prompts
|
|
38
|
+
|
|
39
|
+
The direct arm records a digest of the prompt sent to the provider. The Yoke arm
|
|
40
|
+
records a digest of the PRD and generated configuration submitted to the workflow.
|
|
41
|
+
Those inputs are expanded into story prompts by the separately hashed runtime.
|
|
42
|
+
The manifest labels these different scopes explicitly.
|
|
43
|
+
|
|
44
|
+
The tool does not claim to capture every dynamically expanded prompt, retry
|
|
45
|
+
message, provider system instruction, discovered skill or native context.
|
|
46
|
+
The seed and requirement digests still have to agree across all arms. A prompt
|
|
47
|
+
digest difference is permitted only when the manifest declares it.
|
|
48
|
+
|
|
49
|
+
## Current workflow comparison
|
|
50
|
+
|
|
51
|
+
The comparison still measures one direct Codex session against Yoke's normal
|
|
52
|
+
story execution. Yoke can add worktrees, per-story provider starts, verification
|
|
53
|
+
and commits. These are deliberate workflow differences, not isolated measurements
|
|
54
|
+
of a scheduler or model-call overhead.
|
|
55
|
+
|
|
56
|
+
The existing direct-arm ignore-rules flag is retained and explicitly declared as
|
|
57
|
+
a startup-policy difference. Yoke retains its normal project rule handling.
|
|
58
|
+
Both arms request bare mode and disable native multi-agent delegation. The tool
|
|
59
|
+
does not silently modify global provider configuration or user rules.
|
|
60
|
+
|
|
61
|
+
Model and effort must match across arms unless a different experiment explicitly
|
|
62
|
+
declares those fields as variables. The same applies to actual model identities:
|
|
63
|
+
matching aliases alone do not establish a same-model comparison. An allowed
|
|
64
|
+
difference never makes a missing value valid. Fixture, requirement and acceptance
|
|
65
|
+
differences cannot be waived.
|
|
66
|
+
|
|
67
|
+
Conditions must stay stable within repeated runs of the same arm. A changing
|
|
68
|
+
adaptive model mix therefore needs a separately designed comparison contract;
|
|
69
|
+
this fixed-arm workflow tool does not silently pool such runs.
|
|
70
|
+
|
|
71
|
+
## Analysis and output states
|
|
72
|
+
|
|
73
|
+
Run the analyzer on already recorded result files:
|
|
74
|
+
|
|
75
|
+
node bench/analyze-codex-comparison.mjs path/to/results.json
|
|
76
|
+
|
|
77
|
+
Files may contribute different arms of the same declared comparison. They must
|
|
78
|
+
agree on the manifest and supply every planned arm/repeat exactly once.
|
|
79
|
+
Different comparison identities are never pooled merely because their fixture
|
|
80
|
+
and arm labels match.
|
|
81
|
+
|
|
82
|
+
| State | Meaning |
|
|
83
|
+
| --- | --- |
|
|
84
|
+
| verified | All declared paired runs exist, recorded conditions match the contract, immutable acceptance passes, and input/cache/output telemetry is present |
|
|
85
|
+
| incompatible | Undeclared differences, changing within-arm conditions, conflicting manifests or duplicate measurements |
|
|
86
|
+
| incomplete | Some planned arms or repeats are missing |
|
|
87
|
+
| acceptance-failed | An arm did not pass immutable acceptance; its short failure time cannot count as a speedup |
|
|
88
|
+
| unverified | Missing/unknown provenance or telemetry, an unsupported schema, invalid context or an invalid variable declaration |
|
|
89
|
+
| legacy/unverified | The input predates manifests; its reported provenance and measurements remain readable but are not promoted to a verified comparison |
|
|
90
|
+
|
|
91
|
+
Only verified comparisons produce entries in the top-level groups array.
|
|
92
|
+
Other measurements remain in runs and explicitly labeled diagnostic summaries.
|
|
93
|
+
Missing or partial token measurements remain null. The analyzer does not invent
|
|
94
|
+
USD costs, CPU/memory measurements or savings percentages.
|
|
95
|
+
|
|
96
|
+
A successfully parsed diagnostic report can exit 0 while containing no verified
|
|
97
|
+
comparison. Consumers must inspect comparison status and groups. Malformed JSON,
|
|
98
|
+
impossible cache accounting, invalid timing and empty measurement inputs fail
|
|
99
|
+
the command.
|
|
100
|
+
|
|
101
|
+
## Historical records
|
|
102
|
+
|
|
103
|
+
The analyzer reads both older results arrays and the runs arrays in dated
|
|
104
|
+
aggregate reports. It preserves their reported version and commit under
|
|
105
|
+
legacyReports, marks them legacy/unverified and leaves performance groups empty.
|
|
106
|
+
It no longer hardcodes Yoke 1.19.0 or the September audit commit.
|
|
107
|
+
|
|
108
|
+
The September 29 comparison and September 30 corrected smoke remain historical
|
|
109
|
+
evidence with their original sample sizes and limitations. The latter has no
|
|
110
|
+
contemporaneous baseline and must not be combined with the former to claim a
|
|
111
|
+
new relative speedup.
|
|
112
|
+
|
|
113
|
+
## Measurement limits and authorized execution
|
|
114
|
+
|
|
115
|
+
Manifest verification establishes recorded comparability, not statistical
|
|
116
|
+
confidence, complete environment isolation or general product quality.
|
|
117
|
+
Current fixtures have visible tests and are small. Host load is uncontrolled,
|
|
118
|
+
skills/plugins are not exhaustively audited, and the dependency lockfile does
|
|
119
|
+
not certify every installed dependency or provider-internal setting.
|
|
120
|
+
|
|
121
|
+
Wall time starts after fixture preparation and ends when the runner exits.
|
|
122
|
+
Independent acceptance duration is separate; summaries also report elapsed time
|
|
123
|
+
through that acceptance. Source/environment hashing and fixture setup are
|
|
124
|
+
outside that timed interval.
|
|
125
|
+
|
|
126
|
+
The comparison launcher makes real authenticated provider calls. Run it only
|
|
127
|
+
under the user's authorized experiment scope. It accepts explicit arms, repeats,
|
|
128
|
+
model and effort; its default result root is a short OS-temporary directory
|
|
129
|
+
whose logs are retained. Tooling tests use synthetic records and do not launch
|
|
130
|
+
providers. An existing authorization for exactly two Curiuma development runs
|
|
131
|
+
is not permission for an additional fixture matrix or model-routing study.
|