webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,559 @@
1
+ #!/usr/bin/env node
2
+ import { randomUUID } from 'node:crypto';
3
+ import { appendUsage, readUsage, summarizeUsage, recipeDigest } from './usage.js';
4
+ import { emptyMeta } from './types.js';
5
+ import { measureResult, ExecutionFailure } from './measurement.js';
6
+ import { Command } from 'commander';
7
+ import { gzipSync, gunzipSync } from 'node:zlib';
8
+ import { writeFile, readFile, readdir, mkdir } from 'node:fs/promises';
9
+ import { join, dirname } from 'node:path';
10
+ import { createRequire } from 'node:module';
11
+ import { spawn } from 'node:child_process';
12
+ import { UserError, localPaths, siteName, intentName, publicUrl, verificationWarnings } from './local.js';
13
+ import { buildEngine } from './wiring.js';
14
+ import { formatItems } from './executor/format.js';
15
+ import { learn, formatLearn } from './authoring/learn.js';
16
+ import { teach } from './authoring/teach.js';
17
+ import { PLANS } from '../benchmark/plans.js';
18
+ import { ORACLES } from '../benchmark/oracles.js';
19
+ import { loadTasks, loadGoldens, captureGoldens, runBaseline, runPairs } from './benchmark/runner.js';
20
+ import { summarize, summarizePairs, formatReport, levelOf, failedPairs, excludedBy } from './benchmark/report.js';
21
+ import { screen, formatScreen } from './benchmark/screen.js';
22
+ import { checkRecipes, formatHealth } from './benchmark/health.js';
23
+ import { runVerificationCases, PAGINATION_CASES } from './benchmark/verification-cases.js';
24
+ import { buildCases, runDiscovery, formatDiscovery, DISCOVERY_DOMAINS, HELD_OUT_DOMAINS } from './benchmark/discovery.js';
25
+ import { formatLayers } from './benchmark/verification-matrix.js';
26
+ import { captureDom } from './authoring/snapshot.js';
27
+ import { readUrl, methodStore } from './read.js';
28
+ import { PolitenessLayer, DEFAULT_USER_AGENT } from './net/politeness.js';
29
+ import { visibleRuns, LabelSchema } from './benchmark/labels.js';
30
+ import { scoreSnapshot, formatCorpus } from './benchmark/selector-score.js';
31
+ import { WILD_ORIGINS, StaticSiteResolver } from './sites.js';
32
+ import { startAllFixtures } from '../benchmark/fixtures.js';
33
+ import { fetchTask, listTasks } from './tasks.js';
34
+ import { serveMcp } from './mcp.js';
35
+ const RECIPE_DIR = join(process.cwd(), 'recipes');
36
+ const GOLDEN_DIR = join(process.cwd(), 'benchmark', 'goldens');
37
+ const TASKS = join(process.cwd(), 'benchmark', 'tasks.json');
38
+ const DOM_DIR = join(process.cwd(), 'benchmark', 'dom');
39
+ const tasksFile = (opts) => opts.tasks ?? TASKS;
40
+ function inputFrom(opts) {
41
+ const input = {};
42
+ if (opts.query !== undefined)
43
+ input.query = opts.query;
44
+ if (opts.id !== undefined)
45
+ input.id = opts.id;
46
+ if (opts.page !== undefined) {
47
+ if (!/^[1-9]\d*$/.test(opts.page) || !Number.isSafeInteger(Number(opts.page)))
48
+ throw new UserError('INVALID_INPUT', 'page must be a positive integer');
49
+ input.page = Number(opts.page);
50
+ }
51
+ return input;
52
+ }
53
+ /** `site/intent`, the one handle a saved recipe has. */
54
+ function taskRef(value) {
55
+ const at = value.indexOf('/');
56
+ if (at < 1 || at === value.length - 1)
57
+ throw new UserError('INVALID_INPUT', 'task must be site/intent, for example hn/list');
58
+ return { site: siteName(value.slice(0, at)), intent: intentName(value.slice(at + 1)) };
59
+ }
60
+ /** Reports whether robots.txt permits each wild site's search path. */
61
+ async function robotsStatus(net) {
62
+ const status = {};
63
+ for (const [site, origin] of Object.entries(WILD_ORIGINS)) {
64
+ const plan = PLANS[site]?.search;
65
+ if (!plan)
66
+ continue;
67
+ const url = plan.url(origin, { id: 'probe', site, intent: 'search', input: { query: 'probe' } });
68
+ try {
69
+ status[site] = (await net.isAllowed(url))
70
+ ? 'allowed'
71
+ : 'disallowed for crawlers (user-agent posture, see spec section 6)';
72
+ }
73
+ catch (err) {
74
+ status[site] = `unknown (${err instanceof Error ? err.message : String(err)})`;
75
+ }
76
+ }
77
+ return status;
78
+ }
79
+ const program = new Command().name('webrecipe').description('save how to read a public web page once, then fetch it over plain HTTP')
80
+ .option('--data-dir <path>', 'where recipes are kept (default: WEBRECIPE_DATA_DIR or ~/.webrecipe)')
81
+ .option('--no-log', 'disable local usage logging for this command')
82
+ .exitOverride();
83
+ let usage;
84
+ let usageStarted = 0;
85
+ let usageRoot = '';
86
+ async function writeUsage(event) {
87
+ try {
88
+ await appendUsage(usageRoot, event);
89
+ return true;
90
+ }
91
+ catch (error) {
92
+ console.error(`LOG_WRITE_FAILED: ${error instanceof Error ? error.message : error}`);
93
+ return false;
94
+ }
95
+ }
96
+ program.hook('preAction', async (_program, command) => {
97
+ if (command.parent !== program || !['fetch', 'inspect', 'save', 'read'].includes(command.name()) || !program.opts().log)
98
+ return;
99
+ const opts = command.opts();
100
+ // Positional handles are read leniently here; the action validates them and reports.
101
+ const [site, intent] = command.name() === 'fetch' || command.name() === 'save' ? String(command.args[0] ?? '').split('/') : [];
102
+ const url = command.name() === 'inspect' ? command.args[0] : opts.url;
103
+ usageRoot = localPaths(program.opts().dataDir).root;
104
+ usageStarted = performance.now();
105
+ usage = { version: 1, at: new Date().toISOString(), id: randomUUID(), event: 'start', command: command.name(),
106
+ site, intent, url,
107
+ input: { query: opts.query, id: opts.id, page: opts.page } };
108
+ if (site && intent) {
109
+ try {
110
+ usage.recipeBefore = await recipeDigest(usageRoot, site, intent);
111
+ }
112
+ catch { /* Invalid inputs are reported by the action; no recipe snapshot available. */ }
113
+ }
114
+ await writeUsage(usage);
115
+ });
116
+ program.command('logs').description('summarize local usage; includes failures and unfinished commands')
117
+ .option('--days <days>', 'rolling window', '7')
118
+ .action(async (opts) => {
119
+ const days = Number(opts.days);
120
+ if (!Number.isInteger(days) || days < 1 || days > 365)
121
+ throw new UserError('INVALID_INPUT', 'days must be 1..365');
122
+ const root = localPaths(program.opts().dataDir).root;
123
+ const { events, malformed } = await readUsage(root, days);
124
+ console.log(JSON.stringify({ directory: join(root, 'logs'), days, malformed, ...summarizeUsage(events) }, null, 2));
125
+ });
126
+ program.command('feedback').description('mark a logged fetch as correct or wrong after checking the source')
127
+ .requiredOption('--run <id>')
128
+ .requiredOption('--verdict <verdict>', 'correct or wrong')
129
+ .option('--note <note>')
130
+ .action(async (opts) => {
131
+ if (!['correct', 'wrong'].includes(opts.verdict))
132
+ throw new UserError('INVALID_INPUT', 'verdict must be correct or wrong');
133
+ const root = localPaths(program.opts().dataDir).root;
134
+ const { events } = await readUsage(root, 365);
135
+ if (!events.some(e => e.id === opts.run && e.event === 'finish' && e.command === 'fetch'))
136
+ throw new UserError('INVALID_INPUT', 'run id not found in the last year of logs');
137
+ await appendUsage(root, { version: 1, at: new Date().toISOString(), id: opts.run, event: 'feedback', verdict: opts.verdict, note: opts.note });
138
+ console.log('Feedback saved locally.');
139
+ });
140
+ program.command('setup').description('install the Chromium browser used by inspect, save and browser fallback')
141
+ .action(async () => {
142
+ const require = createRequire(import.meta.url);
143
+ const cli = join(dirname(require.resolve('playwright/package.json')), 'cli.js');
144
+ await new Promise((resolve, reject) => {
145
+ const child = spawn(process.execPath, [cli, 'install', 'chromium'], { stdio: 'inherit' });
146
+ child.on('error', reject);
147
+ child.on('exit', code => code === 0 ? resolve() : reject(new Error(`browser installation failed (${code})`)));
148
+ });
149
+ });
150
+ program.command('list').description('list saved recipes and where they are stored')
151
+ .action(async () => {
152
+ const { root, tasks } = await listTasks(program.opts().dataDir);
153
+ console.log(`Data: ${root}`);
154
+ for (const task of tasks)
155
+ console.log(task);
156
+ if (!tasks.length)
157
+ console.log('No saved recipes yet. Start with: webrecipe inspect <URL>');
158
+ });
159
+ program.command('mcp').description('serve inspect, save, fetch and list as MCP tools over stdio')
160
+ .action(async () => { await serveMcp(program.opts().dataDir); });
161
+ program
162
+ .command('read')
163
+ .description('read any public URL: server HTML when it carries the text, a browser when it does not; remembers which worked per URL shape')
164
+ .requiredOption('--url <url>')
165
+ .option('--format <format>', 'text or json', 'text')
166
+ .action(async (opts) => {
167
+ if (opts.format !== 'text' && opts.format !== 'json')
168
+ throw new UserError('INVALID_INPUT', '--format must be text or json');
169
+ const url = publicUrl(opts.url).toString();
170
+ const net = new PolitenessLayer({ userAgent: DEFAULT_USER_AGENT });
171
+ const result = await readUrl(url, {
172
+ isAllowed: (u) => net.isAllowed(u),
173
+ fetch: (u) => net.fetch(u),
174
+ render: async (u) => (await captureDom(u)).html,
175
+ methods: methodStore(localPaths(program.opts().dataDir).root),
176
+ });
177
+ if (usage)
178
+ Object.assign(usage, { method: result.method, reason: result.reason, textChars: result.text.length });
179
+ if (opts.format === 'json')
180
+ console.log(JSON.stringify({ ok: true, runId: usage?.id, ...result }));
181
+ else
182
+ console.log(`${result.title ? `# ${result.title}\n` : ''}(${result.method}: ${result.reason})\n\n${result.text}`);
183
+ });
184
+ program
185
+ .command('fetch')
186
+ .description('fetch a saved recipe: plain HTTP when the recipe holds, a browser when it does not')
187
+ .argument('<task>', 'site/intent, as saved')
188
+ .option('--query <query>')
189
+ .option('--id <id>')
190
+ .option('--page <page>')
191
+ .option('--json', 'one JSON object on stdout (default: TSV)')
192
+ .option('--no-heal', 'do not recompile the recipe on fallback')
193
+ .action(async (task, opts) => {
194
+ const { site, intent } = taskRef(task);
195
+ const input = inputFrom(opts);
196
+ const { outcome, verification, warnings } = await fetchTask(site, intent, input, { dataDir: program.opts().dataDir, heal: opts.heal, runId: 'cli' });
197
+ if (usage)
198
+ Object.assign(usage, { meta: outcome.meta, items: outcome.items.length, reasons: outcome.reasons, recipeUsed: outcome.recipeUsed, fellBack: outcome.fellBack, blocked: outcome.blocked });
199
+ if (opts.json) {
200
+ console.log(JSON.stringify({ ok: true, runId: usage?.id, items: outcome.items, meta: outcome.meta, verification, warnings }));
201
+ }
202
+ else {
203
+ console.error(`Strategy: ${outcome.meta.strategy} (${levelOf(outcome.meta.strategy)})`);
204
+ console.error(`Verification: ${verification.status}`);
205
+ for (const warning of verificationWarnings(verification))
206
+ console.error(` ${warning}`);
207
+ console.error(`Browser launches: ${outcome.meta.browserLaunches}`);
208
+ console.error(`Elapsed: ${outcome.meta.elapsedMs}ms`);
209
+ console.error(`Requests: ${outcome.meta.networkRequests}`);
210
+ if (outcome.reasons.length)
211
+ console.error(`Fallback: ${outcome.reasons.join('; ')}`);
212
+ console.log(formatItems(outcome.items, 'tsv'));
213
+ }
214
+ });
215
+ program
216
+ .command('inspect')
217
+ .description('open one page in a browser and print the repeated structures and field selectors to choose from')
218
+ .argument('<url>')
219
+ .option('--depth <n>', 'how many item candidates to detail', '10')
220
+ .action(async (url, opts) => {
221
+ publicUrl(url);
222
+ const depth = Number(opts.depth);
223
+ if (!Number.isInteger(depth) || depth < 1 || depth > 50)
224
+ throw new UserError('INVALID_INPUT', 'depth must be between 1 and 50');
225
+ const measured = await measureResult('browser', async () => ({ report: await learn(url, depth), meta: emptyMeta('browser') }));
226
+ if (usage)
227
+ usage.meta = measured.meta;
228
+ console.log(formatLearn(measured.report));
229
+ });
230
+ program
231
+ .command('save')
232
+ .description('save the item selector and fields you chose, then compile an HTTP recipe from them')
233
+ .argument('<task>', 'site/intent, for example hn/list; intent is search, list or detail')
234
+ .requiredOption('--url <url>', 'the page, with any --query/--id/--page value written into it')
235
+ .requiredOption('--items <selector>')
236
+ .requiredOption('--field <name=spec...>', 'repeatable, e.g. --field title="a.title" --field url="a.title@href"')
237
+ .option('--query <query>')
238
+ .option('--id <id>')
239
+ .option('--page <page>')
240
+ .option('--skip-semantic-verification', 'do not spend requests probing what the contract requires; the check stays required and untested')
241
+ .action(async (task, opts) => {
242
+ const { site, intent } = taskRef(task);
243
+ publicUrl(opts.url);
244
+ const paths = localPaths(program.opts().dataDir);
245
+ const fields = Object.fromEntries(opts.field.map((pair) => {
246
+ const at = pair.indexOf('=');
247
+ if (at < 1)
248
+ throw new Error(`--field needs name=spec, got ${JSON.stringify(pair)}`);
249
+ return [pair.slice(0, at), pair.slice(at + 1)];
250
+ }));
251
+ const result = await measureResult('browser', async () => ({ ...await teach({
252
+ site,
253
+ intent,
254
+ url: opts.url,
255
+ input: inputFrom(opts),
256
+ itemSelector: opts.items,
257
+ fields,
258
+ planDir: paths.plans,
259
+ recipeDir: paths.recipes,
260
+ skipSemanticVerification: opts.skipSemanticVerification === true,
261
+ }), meta: emptyMeta('browser') }));
262
+ if (usage)
263
+ Object.assign(usage, { meta: result.meta, recipeStrategy: result.recipe?.strategy.type ?? null, refused: result.refused });
264
+ console.log(`sample: ${JSON.stringify(result.sample)}`);
265
+ console.log(`plan: ${result.planPath}`);
266
+ console.log(`url: ${result.plan.urlTemplate}`);
267
+ console.log(result.recipe === null
268
+ ? `recipe: none — ${result.refused}\n the plan is stored, so fetch falls back to a browser and still answers`
269
+ : `recipe: ${result.recipe.strategy.type} at ${result.recipe.request.path}`);
270
+ const contract = result.plan.verification?.contract;
271
+ if (contract) {
272
+ console.log(`contract: ${contract.required.join(', ')}`);
273
+ // Reported apart from `refused`, which is about compiling a recipe: a
274
+ // verification that came back untested is not a compile failure.
275
+ const honored = result.plan.verification?.evidence.query_honored;
276
+ if (honored)
277
+ console.log(`query_honored: ${honored.status}${honored.reason ? ` — ${honored.reason}` : ''}`);
278
+ else if (contract.required.includes('query_honored'))
279
+ console.log('query_honored: not_tested — probing was skipped');
280
+ }
281
+ });
282
+ const bench = program.command('bench').description('benchmark commands');
283
+ bench
284
+ .command('capture')
285
+ .description('record golden results with the browser; overwrites existing goldens')
286
+ .option('--only <substring>', 'restrict to task ids containing this substring')
287
+ .option('--tasks <path>', 'use a different task file')
288
+ .action(async (opts) => {
289
+ const fixtures = await startAllFixtures();
290
+ const engine = buildEngine({
291
+ recipeDir: RECIPE_DIR, plans: PLANS, heal: false, origins: fixtures.origins,
292
+ });
293
+ try {
294
+ const all = await loadTasks(tasksFile(opts));
295
+ const tasks = opts.only ? all.filter((t) => t.id.includes(opts.only)) : all;
296
+ const goldens = await captureGoldens(tasks, {
297
+ executor: engine.executor, browser: engine.browser, goldenDir: GOLDEN_DIR,
298
+ });
299
+ const hasValues = (g) => g.items.some((item) => Object.values(item).some((v) => v !== null && v !== undefined && v !== ''));
300
+ const expectedEmpty = new Set(tasks.filter((t) => t.expectEmpty).map((t) => t.id));
301
+ const empty = goldens.filter((g) => !expectedEmpty.has(g.taskId) && (g.items.length === 0 || !hasValues(g)));
302
+ console.log(`captured ${goldens.length} goldens`);
303
+ if (empty.length > 0) {
304
+ console.log(`\n${empty.length} golden(s) have no field values — the browser plan is wrong for these:`);
305
+ for (const g of empty)
306
+ console.log(` ${g.taskId}`);
307
+ console.log('\nFix benchmark/plans.ts and recapture. An empty golden passes forever.');
308
+ process.exitCode = 1;
309
+ }
310
+ }
311
+ finally {
312
+ await engine.warm.close();
313
+ await fixtures.close();
314
+ }
315
+ });
316
+ bench
317
+ .command('discovery')
318
+ .description('run the frozen verifier over real sites and record where it and an independent judgement disagree')
319
+ .option('--tasks <paths...>', 'task files to draw the corpus from', ['benchmark/heldout.json', 'benchmark/heldout3.json', 'benchmark/heldout4.json'])
320
+ .option('--out <dir>', 'where to write runs.jsonl, skips.jsonl and the raw evidence')
321
+ .option('--held-out', 'run the untouched domains instead; only after a verifier change')
322
+ .option('--domains <names...>', 'restrict to some of them, for a first pass')
323
+ .action(async (opts) => {
324
+ const all = opts.heldOut === true ? HELD_OUT_DOMAINS : DISCOVERY_DOMAINS;
325
+ const domains = opts.domains === undefined ? all : all.filter((d) => opts.domains.includes(d));
326
+ if (domains.length === 0)
327
+ throw new UserError('INVALID_INPUT', `no such domain in this set; it holds ${all.join(', ')}`);
328
+ const tasks = (await Promise.all(opts.tasks.map(loadTasks))).flat();
329
+ const cases = buildCases(tasks, domains);
330
+ const out = opts.out ?? join(process.cwd(), 'benchmark', 'results', `discovery-${new Date().toISOString().slice(0, 10)}${opts.heldOut === true ? '-heldout' : ''}`);
331
+ console.error(`${cases.length} learned tasks over ${domains.length} domains -> ${out}`);
332
+ const { records, skips } = await runDiscovery(cases, {
333
+ outDir: out,
334
+ onRecord: (r) => console.error(` ${r.finalStatus.padEnd(19)} ${r.judgement.padEnd(13)} ${r.site} ${r.taskId}`),
335
+ onSkip: (s) => console.error(` ${s.reason.padEnd(19)} ${'-'.padEnd(13)} ${s.site} ${s.taskId}`),
336
+ });
337
+ console.log(formatDiscovery(records, skips, cases.length));
338
+ });
339
+ bench
340
+ .command('verification')
341
+ .description("count how often the verifier's verdict and an independent oracle disagree, over the fixture cases")
342
+ .action(async () => {
343
+ console.log('== query cases ==\n');
344
+ console.log(formatLayers(await runVerificationCases(), [
345
+ { label: 'query_honored', required: ['query_honored'] },
346
+ { label: 'query_honored + lexical_query_consistency', required: ['query_honored', 'lexical_query_consistency'] },
347
+ ]));
348
+ // A denominator of its own: these fixtures exercise a page control and the
349
+ // query ones do not, so a pooled rate would be over a population nobody chose.
350
+ console.log('\n\n== pagination cases ==\n');
351
+ console.log(formatLayers(await runVerificationCases(PAGINATION_CASES), [
352
+ { label: 'pagination_honored', required: ['pagination_honored'] },
353
+ ]));
354
+ });
355
+ bench
356
+ .command('screen')
357
+ .description('measure one listing URL before a plan exists: engine page vs browser page, render latency, largest repeated structure')
358
+ .requiredOption('--url <url>')
359
+ .action(async (opts) => {
360
+ console.log(formatScreen(await screen(opts.url)));
361
+ });
362
+ async function readSnapshot(name) {
363
+ return gunzipSync(await readFile(join(DOM_DIR, `${name}.html.gz`))).toString('utf8');
364
+ }
365
+ async function readLabel(name) {
366
+ return LabelSchema.parse(JSON.parse(await readFile(join(DOM_DIR, `${name}.label.json`), 'utf8')));
367
+ }
368
+ bench
369
+ .command('snapshot')
370
+ .description('save one rendered DOM, so a hand-written label keeps its meaning')
371
+ .requiredOption('--url <url>')
372
+ .requiredOption('--as <name>', 'file name to save under, without an extension')
373
+ .action(async (opts) => {
374
+ const { html, capturedAt } = await captureDom(opts.url);
375
+ await mkdir(DOM_DIR, { recursive: true });
376
+ const path = join(DOM_DIR, `${opts.as}.html.gz`);
377
+ const packed = gzipSync(Buffer.from(html, 'utf8'));
378
+ await writeFile(path, packed);
379
+ console.log(`${path} ${(packed.byteLength / 1024).toFixed(1)}KB (${(html.length / 1024).toFixed(1)}KB raw)`);
380
+ console.log(`\nSave the label beside it as ${opts.as}.label.json, filling in items by hand:\n`);
381
+ console.log(JSON.stringify({ snapshot: opts.as, url: opts.url, capturedAt, note: '', identifier: 'text', items: [] }, null, 2));
382
+ console.log(`\nRun \`bench label --snapshot ${opts.as}\` to read the page's text in order.`);
383
+ });
384
+ bench
385
+ .command('label')
386
+ .description("print a snapshot's visible text in document order, for labelling by hand")
387
+ .requiredOption('--snapshot <name>')
388
+ .action(async (opts) => {
389
+ visibleRuns(await readSnapshot(opts.snapshot)).forEach((run, i) => {
390
+ console.log(`${String(i + 1).padStart(4)} ${run}`);
391
+ });
392
+ });
393
+ bench
394
+ .command('selectors')
395
+ .description('rank item-selector candidates against a hand-written label')
396
+ .option('--snapshot <name>', 'one snapshot; omit to run every labelled snapshot')
397
+ .action(async (opts) => {
398
+ const names = opts.snapshot
399
+ ? [opts.snapshot]
400
+ : (await readdir(DOM_DIR).catch(() => []))
401
+ .filter((f) => f.endsWith('.label.json'))
402
+ .map((f) => f.slice(0, -'.label.json'.length))
403
+ .sort();
404
+ if (names.length === 0) {
405
+ console.log('no labelled snapshots in benchmark/dom');
406
+ return;
407
+ }
408
+ const reports = [];
409
+ for (const name of names) {
410
+ reports.push(scoreSnapshot(name, await readSnapshot(name), await readLabel(name)));
411
+ }
412
+ console.log(formatCorpus(reports));
413
+ });
414
+ bench
415
+ .command('health')
416
+ .description('replay the stored recipes and report which are still alive, blocked or broken')
417
+ .option('--tasks <paths...>', 'task files to draw replay inputs from', [TASKS])
418
+ .option('--samples <n>', 'tasks to replay per recipe', '3')
419
+ .action(async (opts) => {
420
+ const tasks = (await Promise.all(opts.tasks.map(loadTasks))).flat();
421
+ const health = await checkRecipes({
422
+ recipeDir: RECIPE_DIR,
423
+ tasks,
424
+ samples: Number(opts.samples),
425
+ net: buildEngine({ recipeDir: RECIPE_DIR, plans: PLANS }).net,
426
+ sites: new StaticSiteResolver(WILD_ORIGINS),
427
+ });
428
+ console.log(formatHealth(health));
429
+ });
430
+ bench
431
+ .command('run')
432
+ .description('run the benchmark and print the controlled/wild report')
433
+ .option('--baseline', 'report the browser baseline instead of the engine')
434
+ .option('--only <substring>', 'restrict to task ids containing this substring')
435
+ .option('--tasks <path>', 'use a different task file')
436
+ .option('--json <path>', 'also write one record per task, so a run split across invocations can be recombined exactly')
437
+ .action(async (opts) => {
438
+ const fixtures = await startAllFixtures();
439
+ const engine = buildEngine({ recipeDir: RECIPE_DIR, plans: PLANS, origins: fixtures.origins });
440
+ try {
441
+ const all = await loadTasks(tasksFile(opts));
442
+ const tasks = opts.only ? all.filter((t) => t.id.includes(opts.only)) : all;
443
+ const goldens = await loadGoldens(GOLDEN_DIR, tasks);
444
+ const deps = { executor: engine.executor, browser: engine.browser, goldenDir: GOLDEN_DIR };
445
+ if (opts.baseline) {
446
+ const baseline = await runBaseline(tasks, goldens, deps);
447
+ // No engine ran, so there is nothing to grade — omit the block rather
448
+ // than print correctness figures this path cannot compute.
449
+ console.log(formatReport(summarize(baseline, baseline), await robotsStatus(engine.net), { correctness: false }));
450
+ const failed = baseline.filter((r) => !r.success);
451
+ if (failed.length > 0) {
452
+ console.log(`failed tasks (${failed.length}/${baseline.length})`);
453
+ for (const f of failed)
454
+ console.log(` ${f.taskId}: ${f.reasons.join('; ') || 'no reason recorded'}`);
455
+ }
456
+ return;
457
+ }
458
+ // Paired, so the engine is not always measured later than its baseline.
459
+ const pairs = await runPairs(tasks, goldens, deps, ORACLES);
460
+ console.log(formatReport(summarizePairs(pairs), await robotsStatus(engine.net)));
461
+ // A median over 50 tasks cannot be rebuilt from five per-site medians, so
462
+ // a run forced into several invocations needs the per-task rows.
463
+ if (opts.json) {
464
+ await writeFile(opts.json, JSON.stringify(pairs.map((p) => ({
465
+ taskId: p.task.id, site: p.task.site, set: p.task.set,
466
+ oracleMode: p.oracle.mode, measurementVersion: 'task-cost-v2',
467
+ engine: { strategy: p.engine.meta.strategy, latencyMs: p.engine.meta.latencyMs,
468
+ elapsedMs: p.engine.meta.elapsedMs,
469
+ unreadResponseBodies: p.engine.meta.unreadResponseBodies,
470
+ browserLaunches: p.engine.meta.browserLaunches,
471
+ politenessWaitMs: p.engine.meta.politenessWaitMs,
472
+ bytesDownloaded: p.engine.meta.bytesDownloaded,
473
+ networkRequests: p.engine.meta.networkRequests,
474
+ agentTokens: p.engine.meta.llmTokens,
475
+ threw: p.engine.threw, blocked: p.engine.blocked,
476
+ items: p.engine.items.length, reasons: p.engine.reasons },
477
+ baseline: { latencyMs: p.baseline.meta.latencyMs,
478
+ elapsedMs: p.baseline.meta.elapsedMs,
479
+ unreadResponseBodies: p.baseline.meta.unreadResponseBodies, threw: p.baseline.threw,
480
+ bytesDownloaded: p.baseline.meta.bytesDownloaded,
481
+ networkRequests: p.baseline.meta.networkRequests,
482
+ agentTokens: p.baseline.meta.llmTokens,
483
+ items: p.baseline.items.length },
484
+ grade: p.grade, aging: p.aging,
485
+ })), null, 1) + '\n');
486
+ }
487
+ // Two different causes, so a healthy browser run whose task simply has no
488
+ // stored golden is never reported as a baseline problem.
489
+ const excludedBaseline = excludedBy(pairs, 'baseline');
490
+ if (excludedBaseline.length > 0) {
491
+ console.log(`excluded (baseline unusable): ${excludedBaseline.length}`);
492
+ for (const e of excludedBaseline)
493
+ console.log(` ${e.taskId}: ${e.reason}`);
494
+ }
495
+ const excludedNoGolden = excludedBy(pairs, 'no-golden');
496
+ if (excludedNoGolden.length > 0) {
497
+ console.log(`excluded (no stored golden): ${excludedNoGolden.length}`);
498
+ for (const e of excludedNoGolden)
499
+ console.log(` ${e.taskId}: ${e.reason}`);
500
+ }
501
+ const aging = pairs.filter((p) => p.aging !== null);
502
+ if (aging.length > 0) {
503
+ console.log(`plan aging (${aging.length}):`);
504
+ for (const p of aging)
505
+ console.log(` ${p.task.id}: ${p.aging}`);
506
+ }
507
+ // Denominator is the usable tasks, matching the rates printed above —
508
+ // an excluded task is unjudged, not a candidate for this list at all.
509
+ const usable = pairs.filter((p) => p.grade.usable);
510
+ const failed = failedPairs(pairs);
511
+ if (failed.length > 0) {
512
+ console.log(`failed tasks (${failed.length}/${usable.length})`);
513
+ for (const f of failed)
514
+ console.log(` ${f.taskId}: ${f.reasons.join('; ') || 'no reason recorded'}`);
515
+ }
516
+ }
517
+ finally {
518
+ await engine.warm.close();
519
+ await fixtures.close();
520
+ }
521
+ });
522
+ try {
523
+ await program.parseAsync(process.argv);
524
+ }
525
+ catch (error) {
526
+ const err = error;
527
+ if (err.exitCode === 0) { /* help/version was printed */ }
528
+ else {
529
+ let cause = err;
530
+ while (cause instanceof Error && !(cause instanceof UserError) && cause.cause)
531
+ cause = cause.cause;
532
+ const code = cause instanceof UserError ? cause.code : err.code?.startsWith('commander.') ? 'INVALID_INPUT' : 'EXECUTION_FAILED';
533
+ if (usage)
534
+ Object.assign(usage, { ok: false, error: { code, message: err.message }, ...(error instanceof ExecutionFailure ? { meta: error.meta } : {}) });
535
+ const payload = { ok: false, runId: usage?.id, error: { code, message: err.message } };
536
+ const jsonRequested = program.commands.some(command => (command.name() === 'fetch' && command.opts().json === true) || (command.name() === 'read' && command.opts().format === 'json'));
537
+ if (jsonRequested)
538
+ console.log(JSON.stringify(payload));
539
+ else
540
+ console.error(`${code}: ${err.message}`);
541
+ process.exitCode = 1;
542
+ }
543
+ }
544
+ finally {
545
+ if (usage) {
546
+ if ('recipeBefore' in usage) {
547
+ try {
548
+ usage.recipeAfter = await recipeDigest(usageRoot, String(usage.site), String(usage.intent));
549
+ usage.recipeChanged = usage.recipeBefore !== usage.recipeAfter;
550
+ }
551
+ catch (error) {
552
+ console.error(`LOG_SNAPSHOT_FAILED: ${String(error)}`);
553
+ }
554
+ }
555
+ const logged = await writeUsage({ ...usage, event: 'finish', at: new Date().toISOString(), ok: usage.ok !== false, wallMs: Math.round(performance.now() - usageStarted) });
556
+ if (logged)
557
+ console.error(`Run log: ${usage.id} (${join(usageRoot, 'logs')})`);
558
+ }
559
+ }