webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,96 @@
1
+ import { stat } from 'node:fs/promises';
2
+ import { join } from 'node:path';
3
+ import { RecipeRegistry } from '../recipes/registry.js';
4
+ import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
5
+ import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
6
+ import { validate } from '../validator/index.js';
7
+ /**
8
+ * Whether the recipes already on disk still work.
9
+ *
10
+ * `bench run` cannot answer this. It pairs every task with a fresh browser
11
+ * baseline, which doubles what the site is asked for, and it grades against
12
+ * goldens captured days earlier, which reads a site editing its content as a
13
+ * recipe that broke. Durability is a narrower question — does the stored
14
+ * recipe still fetch and validate — and it is asked by replaying the recipe
15
+ * and nothing else.
16
+ */
17
+ /** The statuses `src/executor/index.ts` treats as a refusal, for the same reason. */
18
+ const BLOCK_STATUSES = new Set([401, 403, 429, 503]);
19
+ export async function checkRecipes(opts) {
20
+ const registry = new RecipeRegistry(opts.recipeDir);
21
+ const html = new HttpHtmlStrategy(opts.net, opts.sites);
22
+ const json = new HttpJsonStrategy(opts.net, opts.sites);
23
+ const wanted = new Map();
24
+ for (const task of opts.tasks) {
25
+ const key = `${task.site}\u0000${task.intent}`;
26
+ const run = wanted.get(key);
27
+ if (run)
28
+ run.push(task);
29
+ else
30
+ wanted.set(key, [task]);
31
+ }
32
+ const health = [];
33
+ for (const [key, all] of wanted) {
34
+ const [site, intent] = key.split('\u0000');
35
+ const recipe = await registry.load(site, intent);
36
+ // A browser recipe has no HTTP path to outlive, so there is nothing here
37
+ // to measure.
38
+ if (recipe === null || recipe.strategy.type === 'browser')
39
+ continue;
40
+ const strategy = recipe.strategy.type === 'http-json' ? json : html;
41
+ const statuses = [];
42
+ const reasons = [];
43
+ let valid = 0;
44
+ const samples = all.slice(0, opts.samples);
45
+ for (const task of samples) {
46
+ try {
47
+ const result = await strategy.execute(recipe, task);
48
+ if (result.status !== undefined)
49
+ statuses.push(result.status);
50
+ const outcome = validate(recipe, {
51
+ status: result.status ?? recipe.validation.status,
52
+ payload: result.payload,
53
+ items: result.items,
54
+ });
55
+ if (outcome.valid)
56
+ valid += 1;
57
+ else
58
+ reasons.push(`${task.id}: ${outcome.reasons.join('; ')}`);
59
+ }
60
+ catch (err) {
61
+ reasons.push(`${task.id}: ${err instanceof Error ? err.message : String(err)}`);
62
+ }
63
+ }
64
+ // A refusal outranks a mismatch: a site that answered 403 has stopped
65
+ // serving this client, and whatever the recipe would have parsed is moot.
66
+ const verdict = statuses.some((s) => BLOCK_STATUSES.has(s))
67
+ ? 'blocked'
68
+ : valid === samples.length ? 'alive' : 'broken';
69
+ health.push({
70
+ site,
71
+ intent,
72
+ compiledAt: (await stat(join(opts.recipeDir, site, `${intent}.yaml`))).mtime,
73
+ samples: samples.length,
74
+ valid,
75
+ statuses: [...new Set(statuses)],
76
+ verdict,
77
+ reasons,
78
+ });
79
+ }
80
+ return health;
81
+ }
82
+ const age = (from, now) => {
83
+ const hours = Math.round((now.getTime() - from.getTime()) / 3_600_000);
84
+ return hours < 48 ? `${hours}h` : `${Math.round(hours / 24)}d`;
85
+ };
86
+ export function formatHealth(health, now = new Date()) {
87
+ const lines = health.map((h) => ` ${h.verdict.padEnd(8)} ${`${h.site} ${h.intent}`.padEnd(34)} ` +
88
+ `${h.valid}/${h.samples} valid age ${age(h.compiledAt, now).padStart(4)}` +
89
+ `${h.statuses.length > 0 ? ` status ${h.statuses.join(',')}` : ''}`);
90
+ const count = (v) => health.filter((h) => h.verdict === v).length;
91
+ return [
92
+ ...lines,
93
+ '',
94
+ ` alive ${count('alive')} blocked ${count('blocked')} broken ${count('broken')} of ${health.length}`,
95
+ ].join('\n');
96
+ }
@@ -0,0 +1,49 @@
1
+ import { z } from 'zod';
2
+ import { elementsOf, isEl, loadPage, normalize } from '../authoring/candidates.js';
3
+ /**
4
+ * What a human read off the rendered page, and nothing else.
5
+ *
6
+ * There is deliberately nowhere in this schema to put a selector. The whole
7
+ * harness rests on the label being content while the candidate is structure; a
8
+ * label derived from a selector would score that selector against itself, which
9
+ * is the circle the compile-time equivalence check is already stuck in.
10
+ */
11
+ export const LabelSchema = z.object({
12
+ snapshot: z.string().min(1),
13
+ url: z.string().url(),
14
+ capturedAt: z.string().min(1),
15
+ note: z.string().default(''),
16
+ /** Which of the two a human used to write the items down. */
17
+ identifier: z.enum(['text', 'href']),
18
+ items: z.array(z.string().min(1)).min(1),
19
+ });
20
+ /**
21
+ * The page's text in document order, so a human can mark the item boundaries
22
+ * without a selector having proposed them first.
23
+ */
24
+ export function visibleRuns(html) {
25
+ const $ = loadPage(html);
26
+ const runs = [];
27
+ const walk = (node) => {
28
+ for (const child of $(node).contents().toArray()) {
29
+ if (child.type === 'text') {
30
+ const text = normalize($(child).text());
31
+ if (text !== '')
32
+ runs.push(text);
33
+ }
34
+ else if (isEl(child)) {
35
+ // An image speaks through its alt, and on a page of covers that is the
36
+ // only thing the item says.
37
+ const alt = normalize(child.tagName.toLowerCase() === 'img' ? $(child).attr('alt') ?? '' : '');
38
+ if (alt !== '')
39
+ runs.push(alt);
40
+ walk(child);
41
+ }
42
+ }
43
+ };
44
+ const root = elementsOf($, 'body')[0] ?? elementsOf($, 'html')[0];
45
+ if (root === undefined)
46
+ return [];
47
+ walk(root);
48
+ return runs;
49
+ }
@@ -0,0 +1,55 @@
1
+ import { z } from 'zod';
2
+ /**
3
+ * `property-based` is reserved in the spec but deliberately unimplemented, so
4
+ * it is absent here: a task naming it fails at load rather than being quietly
5
+ * graded as something else.
6
+ */
7
+ export const OracleModeSchema = z.enum(['golden', 'paired-live']);
8
+ export const VolatilitySchema = z.enum(['low', 'medium', 'high']);
9
+ // .strict() throughout: a misspelled key (e.g. `field` for `fields`) must fail
10
+ // at load, the same reasoning that kept `property-based` out of the mode enum
11
+ // rather than letting it fall through to a default silently.
12
+ export const OracleSpecSchema = z.object({
13
+ mode: OracleModeSchema.optional(),
14
+ volatility: VolatilitySchema.optional(),
15
+ compare: z.object({
16
+ entities: z.literal('set').optional(),
17
+ ordering: z.enum(['ignore', 'strict']).optional(),
18
+ /** Restricts comparison to the fields that carry meaning. */
19
+ fields: z.array(z.string()).optional(),
20
+ }).strict().optional(),
21
+ /**
22
+ * Opt-in. Declared only where a site is known to be able to ignore its own
23
+ * search input, as arbeitnow.com does. Absent, query semantics is reported
24
+ * as unjudged rather than as a pass.
25
+ */
26
+ semantics: z.object({
27
+ input: z.string(),
28
+ fields: z.array(z.string()),
29
+ match: z.literal('contains-token'),
30
+ minShare: z.number().min(0).max(1),
31
+ }).strict().optional(),
32
+ }).strict();
33
+ /** What every run before this contract effectively used. */
34
+ export const DEFAULT_ORACLE = {
35
+ mode: 'golden',
36
+ volatility: 'low',
37
+ compare: { entities: 'set', ordering: 'strict' },
38
+ };
39
+ /**
40
+ * Merges field by field, so an override may name only `mode` and inherit the
41
+ * rest. Volatility is not uniformly a property of a site: the same host serves
42
+ * a search that rotates in minutes and a detail page that does not.
43
+ */
44
+ export function resolveOracle(site, task) {
45
+ return {
46
+ mode: task?.mode ?? site?.mode ?? DEFAULT_ORACLE.mode,
47
+ volatility: task?.volatility ?? site?.volatility ?? DEFAULT_ORACLE.volatility,
48
+ compare: {
49
+ entities: 'set',
50
+ ordering: task?.compare?.ordering ?? site?.compare?.ordering ?? DEFAULT_ORACLE.compare.ordering,
51
+ fields: task?.compare?.fields ?? site?.compare?.fields,
52
+ },
53
+ semantics: task?.semantics ?? site?.semantics,
54
+ };
55
+ }
@@ -0,0 +1,191 @@
1
+ import { isEquivalent } from './grade.js';
2
+ const LEVELS = {
3
+ 'http-html': 'L0',
4
+ 'http-json': 'L1',
5
+ 'warm-browser': 'L2',
6
+ browser: 'L3',
7
+ };
8
+ export const levelOf = (strategy) => LEVELS[strategy];
9
+ function median(values) {
10
+ if (values.length === 0)
11
+ return 0;
12
+ const sorted = [...values].sort((a, b) => a - b);
13
+ const mid = Math.floor(sorted.length / 2);
14
+ return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
15
+ }
16
+ const share = (rows, predicate) => rows.length === 0 ? 0 : rows.filter(predicate).length / rows.length;
17
+ export function summarize(runs, baseline) {
18
+ const sets = ['controlled', 'wild'];
19
+ return sets.map((set) => {
20
+ const mine = runs.filter((r) => r.set === set);
21
+ const theirs = baseline.filter((r) => r.set === set);
22
+ const medianLatencyMs = median(mine.map((r) => r.meta.latencyMs));
23
+ const baselineMedianLatencyMs = median(theirs.map((r) => r.meta.latencyMs));
24
+ const at = (level) => share(mine, (r) => levelOf(r.meta.strategy) === level);
25
+ const sites = [...new Set(mine.map((r) => r.site))];
26
+ const siteSpeedups = sites.map((site) => {
27
+ const ours = median(mine.filter((r) => r.site === site).map((r) => r.meta.latencyMs));
28
+ const theirsHere = median(theirs.filter((r) => r.site === site).map((r) => r.meta.latencyMs));
29
+ return ours === 0 ? 0 : theirsHere / ours;
30
+ });
31
+ return {
32
+ set,
33
+ tasks: mine.length,
34
+ browserAvoidance: share(mine, (r) => r.meta.browserLaunches === 0 && !r.threw),
35
+ medianSiteSpeedup: median(siteSpeedups),
36
+ levels: { L0: at('L0'), L1: at('L1'), L2: at('L2'), L3: at('L3') },
37
+ browserFree: share(mine, (r) => ['L0', 'L1'].includes(levelOf(r.meta.strategy))),
38
+ fullBrowserAvoidance: share(mine, (r) => r.meta.browserLaunches === 0 && !r.threw),
39
+ blockedTasks: mine.filter((r) => r.blocked).length,
40
+ medianElapsedMs: mine.length > 0 && mine.every((r) => r.meta.elapsedMs !== undefined)
41
+ ? median(mine.map((r) => r.meta.elapsedMs)) : null,
42
+ baselineMedianElapsedMs: theirs.length > 0 && theirs.every((r) => r.meta.elapsedMs !== undefined)
43
+ ? median(theirs.map((r) => r.meta.elapsedMs)) : null,
44
+ medianLatencyMs,
45
+ baselineMedianLatencyMs,
46
+ speedup: medianLatencyMs === 0 ? 0 : baselineMedianLatencyMs / medianLatencyMs,
47
+ successRate: share(mine, (r) => r.success),
48
+ schemaRate: share(mine, (r) => r.schema !== false),
49
+ // This path predates the oracle contract: it never marks a baseline
50
+ // unusable, and it never declares a query-semantics rule.
51
+ usableTasks: mine.length,
52
+ excludedTasks: 0,
53
+ excludedBaselineTasks: 0,
54
+ excludedNoGoldenTasks: 0,
55
+ entityRate: share(mine, (r) => r.schema !== false),
56
+ querySemanticsRate: null,
57
+ querySemanticsJudged: 0,
58
+ orderingRate: (() => {
59
+ const judged = mine.filter((r) => r.ordering !== null && r.ordering !== undefined);
60
+ return judged.length === 0 ? null : share(judged, (r) => r.ordering === true);
61
+ })(),
62
+ medianPolitenessWaitMs: median(mine.map((r) => r.meta.politenessWaitMs)),
63
+ medianBytes: median(mine.map((r) => r.meta.bytesDownloaded)),
64
+ baselineMedianBytes: median(theirs.map((r) => r.meta.bytesDownloaded)),
65
+ medianRequests: median(mine.map((r) => r.meta.networkRequests)),
66
+ baselineMedianRequests: median(theirs.map((r) => r.meta.networkRequests)),
67
+ medianAgentTokens: median(mine.map((r) => r.meta.llmTokens)),
68
+ baselineMedianAgentTokens: median(theirs.map((r) => r.meta.llmTokens)),
69
+ };
70
+ });
71
+ }
72
+ /**
73
+ * Summarises pairs rather than two independent lists.
74
+ *
75
+ * A task whose baseline was unusable leaves the equivalence denominator — the
76
+ * engine cannot be blamed for a browser failure, and counting it as a pass
77
+ * would be worse — but it never leaves the report.
78
+ */
79
+ export function summarizePairs(pairs) {
80
+ const sets = ['controlled', 'wild'];
81
+ return sets.map((set) => {
82
+ const mine = pairs.filter((p) => p.task.set === set);
83
+ const usable = mine.filter((p) => p.grade.usable);
84
+ const engineRuns = mine.map((p) => p.engine);
85
+ const baselineRuns = mine.map((p) => p.baseline);
86
+ const base = summarize(engineRuns, baselineRuns).find((s) => s.set === set);
87
+ const judgedOrdering = usable.filter((p) => p.grade.ordering !== null);
88
+ const judgedQuery = usable.filter((p) => p.grade.query !== null);
89
+ return {
90
+ ...base,
91
+ tasks: mine.length,
92
+ usableTasks: usable.length,
93
+ excludedTasks: mine.length - usable.length,
94
+ // Two different reasons look identical as "not usable": a task with no
95
+ // stored golden had a perfectly healthy browser run, and blaming that on
96
+ // baseline validity is the exact confusion this split exists to end.
97
+ excludedBaselineTasks: mine.filter((p) => p.grade.excludeReason === 'baseline').length,
98
+ excludedNoGoldenTasks: mine.filter((p) => p.grade.excludeReason === 'no-golden').length,
99
+ entityRate: share(usable, (p) => p.grade.entities === true),
100
+ // Denominator is the tasks an oracle actually asked about, so an
101
+ // undeclared rule cannot inflate the rate with unexamined passes.
102
+ querySemanticsJudged: judgedQuery.length,
103
+ querySemanticsRate: judgedQuery.length === 0 ? null : share(judgedQuery, (p) => p.grade.query === true),
104
+ successRate: share(usable, (p) => isEquivalent(p.grade)),
105
+ orderingRate: judgedOrdering.length === 0 ? null : share(judgedOrdering, (p) => p.grade.ordering === true),
106
+ };
107
+ });
108
+ }
109
+ /**
110
+ * The tasks that were judged and did not pass, in the same terms the CLI
111
+ * prints — extracted so the CLI cannot hand-copy `isEquivalent` and drift, the
112
+ * way its old failure filter did.
113
+ */
114
+ export function failedPairs(pairs) {
115
+ return pairs
116
+ .filter((p) => p.grade.usable && !isEquivalent(p.grade))
117
+ .map((p) => ({
118
+ taskId: p.task.id,
119
+ // `grade.reasons` holds only the oracle's verdict text; an engine that
120
+ // timed out or fell back leaves that reason on the run itself, and a
121
+ // printed failure line is useless without it.
122
+ reasons: [...new Set([...p.grade.reasons, ...p.engine.reasons])],
123
+ }));
124
+ }
125
+ /** Tasks left out of the denominator for one specific cause, so the two never print under one heading. */
126
+ export function excludedBy(pairs, cause) {
127
+ return pairs
128
+ .filter((p) => p.grade.excludeReason === cause)
129
+ .map((p) => ({ taskId: p.task.id, reason: p.grade.validityReason }));
130
+ }
131
+ const pct = (n) => `${(n * 100).toFixed(0)}%`;
132
+ /**
133
+ * One cost axis, engine against baseline. Tokens is the first axis where the
134
+ * engine can come out worse, so the direction is read off the numbers rather
135
+ * than assumed to be a saving.
136
+ */
137
+ const ratio = (engine, baseline) => {
138
+ if (engine === 0 || baseline === 0)
139
+ return '—';
140
+ return baseline >= engine ? `${(baseline / engine).toFixed(1)}x less` : `${(engine / baseline).toFixed(1)}x more`;
141
+ };
142
+ export function formatReport(summaries, robots, opts = {}) {
143
+ const { correctness = true } = opts;
144
+ const lines = [];
145
+ for (const s of summaries) {
146
+ if (s.tasks === 0)
147
+ continue;
148
+ lines.push(s.set === 'controlled' ? 'Controlled' : 'Wild');
149
+ lines.push(` Tasks: ${s.tasks}`);
150
+ lines.push(` Browser-free: ${pct(s.browserFree)} (L0 + L1)`);
151
+ lines.push(` Full-browser avoidance: ${pct(s.fullBrowserAvoidance)} (observed zero launches)`);
152
+ if (s.medianElapsedMs !== null) {
153
+ lines.push(` Actual elapsed: ${s.medianElapsedMs}ms median (baseline ${s.baselineMedianElapsedMs ?? 'unavailable'}ms)`);
154
+ if (s.medianElapsedMs > 0 && s.baselineMedianElapsedMs !== null) {
155
+ lines.push(` Actual speedup: ${(s.baselineMedianElapsedMs / s.medianElapsedMs).toFixed(1)}x`);
156
+ }
157
+ }
158
+ lines.push(` Levels: L0 ${pct(s.levels.L0)} L1 ${pct(s.levels.L1)} L2 ${pct(s.levels.L2)} L3 ${pct(s.levels.L3)}`);
159
+ lines.push(` Blocked: ${s.blockedTasks} (site refused the recipe; ran on the browser)`);
160
+ lines.push(` Median latency: ${s.medianLatencyMs}ms (baseline ${s.baselineMedianLatencyMs}ms; legacy attempt timing)`);
161
+ lines.push(` Median speedup: ${s.speedup.toFixed(1)}x (legacy timing)`);
162
+ lines.push(` Median site speedup: ${s.medianSiteSpeedup.toFixed(1)}x`);
163
+ lines.push(` Politeness wait: ${s.medianPolitenessWaitMs}ms median (excluded from latency above)`);
164
+ // Latency is the noisiest of the three cost axes and was the only one
165
+ // reported. Bytes and requests were instrumented from the first commit and
166
+ // discarded on every run until held-out 3.
167
+ const kb = (b) => `${(b / 1024).toFixed(0)}KB`;
168
+ lines.push(` Data fetched: ${kb(s.medianBytes)} median (baseline ${kb(s.baselineMedianBytes)}) ${ratio(s.medianBytes, s.baselineMedianBytes)}`);
169
+ lines.push(` Requests: ${s.medianRequests} median (baseline ${s.baselineMedianRequests})`);
170
+ lines.push(` Agent reads: ${s.medianAgentTokens} tokens median (baseline ${s.baselineMedianAgentTokens}) ${ratio(s.medianAgentTokens, s.baselineMedianAgentTokens)}`);
171
+ // A raw browser-only run (`bench run --baseline`) never grades an engine
172
+ // against anything, so it has none of these figures to report — printing
173
+ // them anyway would be four false labels, not one.
174
+ if (correctness) {
175
+ lines.push(' Correctness');
176
+ lines.push(` Baseline validity: ${s.tasks - s.excludedBaselineTasks}/${s.tasks} usable (${s.excludedBaselineTasks} excluded)`);
177
+ lines.push(` No stored golden: ${s.excludedNoGoldenTasks} excluded`);
178
+ lines.push(` Entity equivalence: ${pct(s.entityRate)}`);
179
+ lines.push(` Query semantics: ${s.querySemanticsRate === null ? 'not declared' : `${pct(s.querySemanticsRate)} of ${s.querySemanticsJudged} judgeable`}`);
180
+ lines.push(` Ordering: ${s.orderingRate === null ? 'not judgeable' : `${pct(s.orderingRate)} where meaningful`}`);
181
+ }
182
+ lines.push('');
183
+ }
184
+ if (Object.keys(robots).length > 0) {
185
+ lines.push('robots.txt status of wild sites');
186
+ for (const [site, status] of Object.entries(robots))
187
+ lines.push(` ${site}: ${status}`);
188
+ lines.push('');
189
+ }
190
+ return lines.join('\n');
191
+ }
@@ -0,0 +1,201 @@
1
+ import { ExecutionFailure } from '../measurement.js';
2
+ import { mkdir, readFile, writeFile } from 'node:fs/promises';
3
+ import { join } from 'node:path';
4
+ import { z } from 'zod';
5
+ import { gradeGolden } from './golden.js';
6
+ import { gradePair, planAging, querySemantics } from './grade.js';
7
+ import { OracleSpecSchema, resolveOracle } from './oracle.js';
8
+ const BenchTaskSchema = z.object({
9
+ id: z.string(),
10
+ set: z.enum(['controlled', 'wild']),
11
+ site: z.string(),
12
+ intent: z.enum(['search', 'list', 'detail']),
13
+ input: z.record(z.union([z.string(), z.number()])),
14
+ volatile: z.array(z.string()).optional(),
15
+ /** A search that legitimately returns nothing; an empty golden is correct here. */
16
+ expectEmpty: z.boolean().optional(),
17
+ oracle: OracleSpecSchema.optional(),
18
+ });
19
+ export async function loadTasks(path) {
20
+ const raw = JSON.parse(await readFile(path, 'utf8'));
21
+ return z.array(BenchTaskSchema).parse(raw);
22
+ }
23
+ function goldenPath(dir, taskId) {
24
+ return join(dir, `${taskId}.json`);
25
+ }
26
+ function failedRun(task, error) {
27
+ const reason = message(error);
28
+ return {
29
+ taskId: task.id,
30
+ site: task.site,
31
+ set: task.set,
32
+ meta: error instanceof ExecutionFailure ? error.meta : {
33
+ strategy: 'browser', latencyMs: 0, browserLaunches: 0,
34
+ pageNavigations: 0, networkRequests: 0, bytesDownloaded: 0, llmTokens: 0, politenessWaitMs: 0,
35
+ },
36
+ success: false,
37
+ reasons: [reason],
38
+ items: [],
39
+ threw: true,
40
+ blocked: false,
41
+ };
42
+ }
43
+ const message = (err) => (err instanceof Error ? err.message : String(err));
44
+ /** Explicit, never implicit: a golden is only written when this is called. */
45
+ export async function captureGoldens(tasks, deps) {
46
+ await mkdir(deps.goldenDir, { recursive: true });
47
+ const goldens = [];
48
+ for (const task of tasks) {
49
+ const result = await deps.browser.execute({}, task);
50
+ const golden = { taskId: task.id, capturedAt: new Date().toISOString(), items: result.items };
51
+ await writeFile(goldenPath(deps.goldenDir, task.id), JSON.stringify(golden, null, 2), 'utf8');
52
+ goldens.push(golden);
53
+ }
54
+ return goldens;
55
+ }
56
+ export async function loadGoldens(dir, tasks) {
57
+ const goldens = [];
58
+ for (const task of tasks) {
59
+ try {
60
+ goldens.push(JSON.parse(await readFile(goldenPath(dir, task.id), 'utf8')));
61
+ }
62
+ catch {
63
+ // A missing golden means this task has never been captured; skip it.
64
+ }
65
+ }
66
+ return goldens;
67
+ }
68
+ export async function runBaseline(tasks, goldens, deps) {
69
+ const byId = new Map(goldens.map((g) => [g.taskId, g]));
70
+ const runs = [];
71
+ for (const task of tasks) {
72
+ try {
73
+ const result = await deps.browser.execute({}, task);
74
+ const golden = byId.get(task.id);
75
+ const grade = golden
76
+ ? gradeGolden(golden, result.items, task.volatile)
77
+ : { schema: true, semantic: true, ordering: null, reasons: [] };
78
+ runs.push({
79
+ taskId: task.id, site: task.site, set: task.set, meta: result.meta,
80
+ success: grade.schema && grade.semantic, schema: grade.schema, ordering: grade.ordering,
81
+ reasons: grade.reasons,
82
+ items: result.items, threw: false, blocked: false,
83
+ });
84
+ }
85
+ catch (err) {
86
+ runs.push(failedRun(task, err));
87
+ }
88
+ }
89
+ return runs;
90
+ }
91
+ async function runBenchmark(tasks, goldens, deps) {
92
+ const byId = new Map(goldens.map((g) => [g.taskId, g]));
93
+ const runs = [];
94
+ for (const task of tasks) {
95
+ try {
96
+ const outcome = await deps.executor.run(task);
97
+ const golden = byId.get(task.id);
98
+ const grade = golden
99
+ ? gradeGolden(golden, outcome.items, task.volatile)
100
+ : { schema: true, semantic: true, ordering: null, reasons: [] };
101
+ runs.push({
102
+ taskId: task.id, site: task.site, set: task.set, meta: outcome.meta,
103
+ success: grade.schema && grade.semantic, schema: grade.schema, ordering: grade.ordering,
104
+ reasons: [...outcome.reasons, ...grade.reasons],
105
+ items: outcome.items, threw: false, blocked: outcome.blocked,
106
+ });
107
+ }
108
+ catch (err) {
109
+ runs.push(failedRun(task, err));
110
+ }
111
+ }
112
+ return runs;
113
+ }
114
+ /** An engine that threw produced no items to grade; scoring it against an
115
+ * empty reference (an `expectEmpty` baseline, an empty golden) would read as a
116
+ * pass, which is worse than the failure it stands in for. Decided before
117
+ * either mode's layer two runs, so the two never diverge on it. */
118
+ function engineException(engine) {
119
+ return {
120
+ entities: false, query: null, ordering: null,
121
+ reasons: engine.reasons.length > 0 ? engine.reasons : ['engine threw'],
122
+ };
123
+ }
124
+ /**
125
+ * Grades by mode, because the two modes ask different questions.
126
+ *
127
+ * `golden` asks whether the engine matches a stored capture. A live browser
128
+ * that times out says nothing about that, so it must not gate it — routing
129
+ * every mode through `gradePair` would have excluded a task whose engine
130
+ * matched its golden exactly, on the strength of an unrelated timeout. The
131
+ * baseline is still run, because the speedup figure needs it.
132
+ *
133
+ * `paired-live` asks whether the engine matches the browser beside it, so there
134
+ * the baseline's validity is exactly the right gate.
135
+ */
136
+ export function gradeByOracle(oracle, stored, baseline, engine, task) {
137
+ if (oracle.mode === 'paired-live') {
138
+ const grade = gradePair(baseline.items, engine.items, oracle, {
139
+ threw: baseline.threw,
140
+ expectEmpty: task.expectEmpty === true,
141
+ input: task.input,
142
+ });
143
+ return grade.usable && engine.threw ? { ...grade, ...engineException(engine) } : grade;
144
+ }
145
+ if (stored === undefined) {
146
+ return {
147
+ usable: false, validityReason: 'no stored golden for this task', excludeReason: 'no-golden',
148
+ entities: null, query: null, ordering: null,
149
+ reasons: ['no stored golden for this task'],
150
+ };
151
+ }
152
+ if (engine.threw) {
153
+ return { usable: true, validityReason: null, excludeReason: null, ...engineException(engine) };
154
+ }
155
+ const graded = gradeGolden(stored, engine.items, task.volatile, oracle.compare.fields);
156
+ const query = querySemantics(engine.items, task.input, oracle);
157
+ return {
158
+ usable: true,
159
+ validityReason: null,
160
+ excludeReason: null,
161
+ entities: graded.schema && graded.semantic,
162
+ query: query.match,
163
+ ordering: oracle.compare.ordering === 'ignore' ? null : graded.ordering,
164
+ reasons: [...graded.reasons, ...query.reasons],
165
+ };
166
+ }
167
+ /**
168
+ * Runs both sides of each task before moving to the next, alternating which
169
+ * goes first.
170
+ *
171
+ * Running the whole baseline and then the whole engine put every engine
172
+ * measurement later in time than its baseline, which on a job board cost 25
173
+ * points of apparent correctness. Pairing removes that; alternating removes the
174
+ * residual bias of one side always going first. Comparison happens once both
175
+ * halves are in hand, so order within a pair does not decide which is the
176
+ * reference.
177
+ */
178
+ export async function runPairs(tasks, goldens, deps, oracles) {
179
+ const byId = new Map(goldens.map((g) => [g.taskId, g]));
180
+ const pairs = [];
181
+ for (const [index, task] of tasks.entries()) {
182
+ const oracle = resolveOracle(OracleSpecSchema.optional().parse(oracles[task.site]), task.oracle);
183
+ const engineFirst = index % 2 === 1;
184
+ let baseline;
185
+ let engine;
186
+ if (engineFirst) {
187
+ engine = (await runBenchmark([task], goldens, deps))[0];
188
+ baseline = (await runBaseline([task], goldens, deps))[0];
189
+ }
190
+ else {
191
+ baseline = (await runBaseline([task], goldens, deps))[0];
192
+ engine = (await runBenchmark([task], goldens, deps))[0];
193
+ }
194
+ const grade = gradeByOracle(oracle, byId.get(task.id), baseline, engine, task);
195
+ const aging = oracle.mode === 'paired-live'
196
+ ? planAging(byId.get(task.id)?.items, baseline.items, oracle)
197
+ : null;
198
+ pairs.push({ task, oracle, baseline, engine, grade, aging });
199
+ }
200
+ return pairs;
201
+ }