webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,355 @@
1
+ import { mkdtemp, mkdir, writeFile, appendFile } from 'node:fs/promises';
2
+ import { tmpdir } from 'node:os';
3
+ import { join } from 'node:path';
4
+ import { load } from 'cheerio';
5
+ import { teach } from '../authoring/teach.js';
6
+ import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
7
+ import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
8
+ import { PolitenessLayer } from '../net/politeness.js';
9
+ import { StaticSiteResolver, WILD_ORIGINS } from '../sites.js';
10
+ import { verifyReadable, UserError } from '../local.js';
11
+ import { PLANS } from '../../benchmark/plans.js';
12
+ /**
13
+ * What the frozen verifier says about real sites, beside what an independent
14
+ * account says the answer was.
15
+ *
16
+ * The verifier is not modified while this runs. A failure found on the first
17
+ * site and fixed before the last is a batch whose halves came from two
18
+ * different systems, and the first case is the one a fix overfits to.
19
+ */
20
+ export const BASELINE = '7f51ca0';
21
+ /** At most four, plus every task that expects an empty result: those carry a
22
+ * deterministic oracle and are worth more than another ordinary query. */
23
+ const UNSEEN_PER_CASE = 4;
24
+ export function buildCases(tasks, domains) {
25
+ const groups = new Map();
26
+ for (const task of tasks) {
27
+ if (!domains.includes(task.site))
28
+ continue;
29
+ const key = `${task.site}\u0000${task.intent}`;
30
+ groups.set(key, [...(groups.get(key) ?? []), task]);
31
+ }
32
+ return [...groups.values()].map((all) => {
33
+ const [taught, ...rest] = all;
34
+ const empty = rest.filter((t) => t.expectEmpty === true);
35
+ const ordinary = rest.filter((t) => t.expectEmpty !== true).slice(0, UNSEEN_PER_CASE - empty.length);
36
+ return { site: taught.site, intent: taught.intent, taught: taught, unseen: [...ordinary, ...empty] };
37
+ });
38
+ }
39
+ /** Reads the signals each verifier already records; no verifier was changed. */
40
+ export function classifyNotTested(evidence) {
41
+ if (evidence === undefined)
42
+ return 'insufficient_evidence';
43
+ const declined = (evidence.probes ?? []).find((p) => p.unavailable !== undefined)?.unavailable ?? '';
44
+ if (/429|retry later/i.test(declined))
45
+ return 'rate_limited';
46
+ if (declined !== '')
47
+ return 'transport_failure';
48
+ const s = evidence.signals ?? {};
49
+ if (s.stable === false)
50
+ return 'volatile';
51
+ if (s.applicable === false)
52
+ return 'no_visible_lexical_match';
53
+ if (s.consistent === false)
54
+ return 'lexical_inconsistent';
55
+ if (s.agreed === false)
56
+ return 'browser_disagreement';
57
+ if (s.moved === false)
58
+ return 'page_control_ambiguous';
59
+ if ((evidence.probes ?? []).some((p) => p.role === 'alternate' && p.items === 0))
60
+ return 'empty_alternate_page';
61
+ if (!(evidence.probes ?? []).some((p) => p.role === 'contrast' || p.role === 'alternate'))
62
+ return 'insufficient_evidence';
63
+ return 'unclassified';
64
+ }
65
+ const WORD = /^[a-z][a-z0-9-]{3,}$/;
66
+ const tokens = (text) => [...new Set(text.toLowerCase().split(/[^a-z0-9-]+/i).filter((t) => WORD.test(t)))];
67
+ const carriesId = (items, id) => {
68
+ const wanted = id.toLowerCase();
69
+ return items.some((item) => Object.values(item).some((v) => String(v ?? '').toLowerCase().includes(wanted)));
70
+ };
71
+ /**
72
+ * Whether the id rule can be applied to this plan at all.
73
+ *
74
+ * Decided from the taught observation, which is the browser's reading of the
75
+ * url the task names and so is the right entity by the task's own definition.
76
+ * If the id cannot be seen even there, this plan does not extract anything that
77
+ * carries one and the rule tells us nothing about any other input.
78
+ *
79
+ * This answers only "can this oracle be used here". It is never the answer to
80
+ * "was this result right", which is what keeps it from grading itself.
81
+ */
82
+ export function entityIdApplicable(taught, taughtItems) {
83
+ const type = 'entity_id';
84
+ if (taught.intent !== 'detail')
85
+ return { type, applicable: false, reason: 'not a detail lookup' };
86
+ const id = taught.input.id;
87
+ if (id === undefined || String(id) === '')
88
+ return { type, applicable: false, reason: 'the task names no entity id' };
89
+ if (taughtItems === null || taughtItems.length === 0)
90
+ return { type, applicable: false, reason: 'the taught observation could not be read' };
91
+ return carriesId(taughtItems, String(id))
92
+ ? { type, applicable: true, reason: `the taught output carries the entity id ${id}` }
93
+ : { type, applicable: false, reason: 'requested entity id is not observable in the taught output' };
94
+ }
95
+ /**
96
+ * T1. A deterministic reading of the task itself, owing nothing to the probes.
97
+ */
98
+ export function judgeByTask(task, items, role, entityId) {
99
+ if (task.expectEmpty === true) {
100
+ const oracle = { type: 'expect_empty', applicable: true, reason: 'the task declares this query matches nothing' };
101
+ return items.length === 0
102
+ ? { judgement: 'correct', source: 'expect_empty', reason: 'a query declared to match nothing returned nothing', oracle }
103
+ : { judgement: 'wrong', source: 'expect_empty', reason: `a query declared to match nothing returned ${items.length} rows`, oracle };
104
+ }
105
+ if (!entityId.applicable)
106
+ return { judgement: 'indeterminate', source: 'none', reason: entityId.reason, oracle: entityId };
107
+ // The input the gate was calibrated on cannot also be judged by it.
108
+ if (role === 'taught') {
109
+ return {
110
+ judgement: 'indeterminate', source: 'none',
111
+ reason: 'this is the observation the id oracle was calibrated on, so it cannot also answer for it',
112
+ oracle: { ...entityId, applicable: false, reason: 'calibration input' },
113
+ };
114
+ }
115
+ const id = String(task.input.id);
116
+ if (items.length === 0) {
117
+ return { judgement: 'indeterminate', source: 'none', reason: 'nothing returned to check the id against', oracle: entityId };
118
+ }
119
+ return carriesId(items, id)
120
+ ? { judgement: 'correct', source: 'entity_id', reason: `the requested id ${id} appears in the answer`, oracle: entityId }
121
+ : { judgement: 'wrong', source: 'entity_id', reason: `the requested id ${id} appears nowhere in the answer`, oracle: entityId };
122
+ }
123
+ /**
124
+ * T2. One-sided on purpose.
125
+ *
126
+ * A row that does not appear on the page it links to is a row describing
127
+ * something else, which is evidence the answer is wrong. A row that does appear
128
+ * there proves only that the row and its own link agree: move a whole row to
129
+ * the wrong entity — title, url and all — and this check still passes it. So a
130
+ * match settles nothing and goes to the next tier.
131
+ */
132
+ export async function detailMismatch(items, net, origin, sample = 2) {
133
+ let requests = 0;
134
+ for (const item of items.slice(0, sample)) {
135
+ const href = String(item.url ?? '');
136
+ const title = String(item.title ?? '');
137
+ if (href === '' || title === '')
138
+ continue;
139
+ const wanted = tokens(title);
140
+ if (wanted.length === 0)
141
+ continue;
142
+ let url;
143
+ try {
144
+ url = new URL(href, origin).toString();
145
+ }
146
+ catch {
147
+ continue;
148
+ }
149
+ if (new URL(url).origin !== origin)
150
+ continue;
151
+ if (!(await net.isAllowed(url)))
152
+ continue;
153
+ let text;
154
+ try {
155
+ requests += 1;
156
+ const res = await net.fetch(url);
157
+ if (res.status !== 200)
158
+ continue;
159
+ text = load(res.body).text().toLowerCase();
160
+ }
161
+ catch {
162
+ continue;
163
+ }
164
+ // A shell that rendered nothing cannot disagree with anything.
165
+ if (text.length < 500)
166
+ continue;
167
+ if (!wanted.some((t) => text.includes(t))) {
168
+ return { wrong: true, requests, reason: `the row "${title}" shares no word with the page at ${href} that it links to` };
169
+ }
170
+ }
171
+ return { wrong: false, requests, reason: '' };
172
+ }
173
+ export const DISCOVERY_DOMAINS = [
174
+ 'hex.pm', 'docs.rs', 'jsr.io', 'bandcamp.com',
175
+ 'meta.discourse.org', 'lemmy.world', 'openlibrary.org', 'dev.to',
176
+ ];
177
+ /** Untouched by this batch, and run once when a changed verifier is first evaluated. */
178
+ export const HELD_OUT_DOMAINS = ['npmjs.com', 'musicbrainz.org', 'mastodon.social', 'itch.io'];
179
+ export async function runDiscovery(cases, opts) {
180
+ await mkdir(join(opts.outDir, 'evidence'), { recursive: true });
181
+ const records = [];
182
+ const skips = [];
183
+ const net = new PolitenessLayer({ minIntervalMs: opts.minIntervalMs });
184
+ const skip = async (s) => {
185
+ skips.push(s);
186
+ opts.onSkip?.(s);
187
+ await appendFile(join(opts.outDir, 'skips.jsonl'), `${JSON.stringify(s)}\n`);
188
+ };
189
+ const keep = async (r) => {
190
+ records.push(r);
191
+ opts.onRecord?.(r);
192
+ await appendFile(join(opts.outDir, 'runs.jsonl'), `${JSON.stringify(r)}\n`);
193
+ };
194
+ for (const one of cases) {
195
+ const origin = WILD_ORIGINS[one.site];
196
+ const plan = PLANS[one.site]?.[one.intent];
197
+ if (origin === undefined || plan === undefined) {
198
+ await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'no_plan', detail: 'no hand-written plan for this site and intent' });
199
+ continue;
200
+ }
201
+ const taughtUrl = plan.url(origin, one.taught);
202
+ if (!(await net.isAllowed(taughtUrl))) {
203
+ await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'skipped_by_robots', detail: taughtUrl });
204
+ continue;
205
+ }
206
+ const planDir = await mkdtemp(join(tmpdir(), 'disc-plans-'));
207
+ const recipeDir = await mkdtemp(join(tmpdir(), 'disc-recipes-'));
208
+ const startedTeach = performance.now();
209
+ let taught;
210
+ try {
211
+ taught = await teach({
212
+ site: one.site, intent: one.intent, url: taughtUrl, input: one.taught.input,
213
+ itemSelector: plan.itemSelector, fields: plan.fields,
214
+ planDir, recipeDir, minIntervalMs: opts.minIntervalMs,
215
+ });
216
+ }
217
+ catch (error) {
218
+ await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'teach_failed', detail: error instanceof Error ? error.message : String(error) });
219
+ continue;
220
+ }
221
+ const teachMs = Math.round(performance.now() - startedTeach);
222
+ const verification = taught.plan.verification;
223
+ const contract = verification?.contract.required ?? [];
224
+ const notTestedReasons = {};
225
+ for (const check of contract) {
226
+ const evidence = verification?.evidence[check];
227
+ if (check === 'non_empty' || check === 'required_fields')
228
+ continue;
229
+ if (evidence?.status === 'passed' || evidence?.status === 'failed')
230
+ continue;
231
+ notTestedReasons[check] = classifyNotTested(evidence);
232
+ }
233
+ await writeFile(join(opts.outDir, 'evidence', `${one.site}-${one.intent}.json`), `${JSON.stringify({ baseline: BASELINE, site: one.site, intent: one.intent, taughtUrl, plan: taught.plan, refused: taught.refused, sample: taught.sample }, null, 2)}\n`);
234
+ const sites = new StaticSiteResolver({ [one.site]: origin });
235
+ const fields = Object.keys(plan.fields);
236
+ // Calibrated once per plan, from the browser's reading of the url the task
237
+ // names. It decides whether the id rule applies here, never what the answer was.
238
+ const entityId = entityIdApplicable(one.taught, taught.sample);
239
+ for (const [role, task] of [['taught', one.taught], ...one.unseen.map((t) => ['unseen', t])]) {
240
+ const url = plan.url(origin, task);
241
+ if (!(await net.isAllowed(url))) {
242
+ await skip({ baseline: BASELINE, site: one.site, taskId: task.id, reason: 'skipped_by_robots', detail: url });
243
+ continue;
244
+ }
245
+ const started = performance.now();
246
+ let items = null;
247
+ let error;
248
+ let requests = 0;
249
+ if (taught.recipe !== null) {
250
+ const strategy = taught.recipe.output.type === 'json' ? new HttpJsonStrategy(net, sites) : new HttpHtmlStrategy(net, sites);
251
+ try {
252
+ requests += 1;
253
+ const result = await strategy.execute(taught.recipe, { id: task.id, site: one.site, intent: one.intent, input: task.input });
254
+ items = result.status !== undefined && result.status !== taught.recipe.validation.status ? null : result.items;
255
+ if (items === null)
256
+ error = `status ${result.status}`;
257
+ }
258
+ catch (err) {
259
+ error = err instanceof Error ? err.message : String(err);
260
+ }
261
+ }
262
+ else {
263
+ error = `no recipe: ${taught.refused ?? 'unknown'}`;
264
+ }
265
+ let finalStatus = 'error';
266
+ let checks = {};
267
+ if (items !== null) {
268
+ try {
269
+ const read = verifyReadable(items, fields, verification);
270
+ finalStatus = read.status;
271
+ checks = read.checks;
272
+ }
273
+ catch (err) {
274
+ finalStatus = 'unverified';
275
+ if (err instanceof UserError)
276
+ error = err.message;
277
+ }
278
+ }
279
+ let { judgement, source, reason, oracle } = items === null
280
+ ? {
281
+ judgement: 'indeterminate', source: 'none',
282
+ reason: error ?? 'no answer',
283
+ oracle: { type: 'none', applicable: false, reason: 'no answer to judge' },
284
+ }
285
+ : judgeByTask(task, items, role, entityId);
286
+ if (judgement === 'indeterminate' && items !== null && items.length > 0 && one.intent !== 'detail') {
287
+ const probe = await detailMismatch(items, net, origin);
288
+ requests += probe.requests;
289
+ if (probe.wrong) {
290
+ judgement = 'wrong';
291
+ source = 'detail_mismatch';
292
+ reason = probe.reason;
293
+ oracle = { type: 'detail_mismatch', applicable: true, reason: 'a returned row could be checked against the page it links to' };
294
+ }
295
+ }
296
+ // A person settles what reached verified, and a sample of what did not.
297
+ const needsManual = judgement === 'indeterminate' && (finalStatus === 'verified' || finalStatus === 'partially_verified');
298
+ await keep({
299
+ baseline: BASELINE, cached: false, site: one.site, taskId: task.id, intent: one.intent,
300
+ inputRole: role, input: task.input, contract, checks, finalStatus, notTestedReasons,
301
+ judgement, judgementSource: source, judgementReason: reason, oracle, needsManual,
302
+ items: items ?? [], itemCount: items?.length ?? 0,
303
+ cost: { httpRequests: requests, browserRuns: role === 'taught' ? 3 : 0, elapsedMs: Math.round(performance.now() - started) + (role === 'taught' ? teachMs : 0) },
304
+ ...(error === undefined ? {} : { error }),
305
+ });
306
+ }
307
+ }
308
+ return { records, skips };
309
+ }
310
+ export function formatDiscovery(records, skips, eligible) {
311
+ const lines = [];
312
+ const judged = (j) => (r) => r.judgement === j;
313
+ const status = (s) => (r) => r.finalStatus === s;
314
+ const count = (f) => records.filter(f).length;
315
+ const both = (a, b) => count((r) => a(r) && b(r));
316
+ const domains = new Set(records.map((r) => r.site));
317
+ lines.push(`baseline: ${BASELINE} cached: false (every request hit the origin)`);
318
+ lines.push(`domains: ${domains.size} learned tasks: ${eligible} evaluated inputs: ${records.length}`);
319
+ const byReason = new Map();
320
+ for (const s of skips)
321
+ byReason.set(s.reason, (byReason.get(s.reason) ?? 0) + 1);
322
+ lines.push(`skipped: ${skips.length}${skips.length === 0 ? '' : ` (${[...byReason].map(([k, v]) => `${k} ${v}`).join(', ')})`}`);
323
+ lines.push('');
324
+ lines.push(' correct wrong indeterminate');
325
+ for (const s of ['verified', 'partially_verified', 'structural', 'unverified', 'error']) {
326
+ const row = records.filter(status(s));
327
+ if (row.length === 0)
328
+ continue;
329
+ lines.push(` ${s.padEnd(20)}${String(both(status(s), judged('correct'))).padStart(5)}${String(both(status(s), judged('wrong'))).padStart(8)}${String(both(status(s), judged('indeterminate'))).padStart(14)}`);
330
+ }
331
+ lines.push('');
332
+ const falseSuccess = records.filter((r) => r.finalStatus === 'verified' && r.judgement === 'wrong');
333
+ lines.push(`false success (verified + wrong): ${falseSuccess.length}`);
334
+ for (const r of falseSuccess)
335
+ lines.push(` ${r.site} ${r.taskId} ${JSON.stringify(r.input)} — ${r.judgementReason}`);
336
+ lines.push('');
337
+ lines.push('correct but not verified, by reason:');
338
+ const reasons = new Map();
339
+ for (const r of records) {
340
+ if (r.judgement !== 'correct' || r.finalStatus === 'verified')
341
+ continue;
342
+ for (const reason of Object.values(r.notTestedReasons))
343
+ reasons.set(reason, (reasons.get(reason) ?? 0) + 1);
344
+ if (Object.keys(r.notTestedReasons).length === 0)
345
+ reasons.set('structural_only', (reasons.get('structural_only') ?? 0) + 1);
346
+ }
347
+ for (const [reason, n] of [...reasons].sort((a, b) => b[1] - a[1]))
348
+ lines.push(` ${reason.padEnd(26)} ${n}`);
349
+ lines.push('');
350
+ const queue = records.filter((r) => r.needsManual);
351
+ lines.push(`awaiting manual judgement: ${queue.length}`);
352
+ for (const r of queue.slice(0, 40))
353
+ lines.push(` [${r.finalStatus}] ${r.site} ${r.taskId} ${JSON.stringify(r.input)} items=${r.itemCount}`);
354
+ return lines.join('\n');
355
+ }
@@ -0,0 +1,95 @@
1
+ export function diffGolden(golden, actual, volatile = []) {
2
+ const reasons = [];
3
+ if (actual.length !== golden.items.length) {
4
+ reasons.push(`item count ${actual.length} !== golden ${golden.items.length}`);
5
+ return { match: false, reasons };
6
+ }
7
+ const volatileSet = new Set(volatile);
8
+ golden.items.forEach((expected, index) => {
9
+ const got = actual[index];
10
+ if (!got) {
11
+ reasons.push(`item ${index} missing`);
12
+ return;
13
+ }
14
+ for (const [field, expectedValue] of Object.entries(expected)) {
15
+ const gotValue = got[field];
16
+ if (volatileSet.has(field)) {
17
+ if (!(field in got) || gotValue === null || gotValue === undefined) {
18
+ reasons.push(`item ${index}: volatile field "${field}" absent`);
19
+ }
20
+ else if (typeof gotValue !== typeof expectedValue) {
21
+ reasons.push(`item ${index}: volatile field "${field}" changed type`);
22
+ }
23
+ continue;
24
+ }
25
+ if (gotValue !== expectedValue) {
26
+ reasons.push(`item ${index}: "${field}" ${JSON.stringify(gotValue)} !== ${JSON.stringify(expectedValue)}`);
27
+ }
28
+ }
29
+ });
30
+ return { match: reasons.length === 0, reasons };
31
+ }
32
+ const key = (item, volatile, fields) => JSON.stringify((fields ?? Object.keys(item)).filter((k) => !volatile.has(k)).sort().map((k) => [k, item[k]]));
33
+ /**
34
+ * Three separate questions, because collapsing them mixes engine failure with
35
+ * website nondeterminism. flathub and pkg.go.dev reorder equally-ranked results
36
+ * between one request and the next, which failed even the browser baseline
37
+ * against its own golden — that is the site being unstable, not the recipe
38
+ * being wrong, and it should not be reported as the same thing.
39
+ *
40
+ * `fields`, when given, restricts both schema and semantic comparison to that
41
+ * set — the same restriction `oracle.compare.fields` applies under
42
+ * `paired-live`, so a column the oracle does not score cannot fail a golden
43
+ * task either.
44
+ */
45
+ export function gradeGolden(golden, actual, volatile = [], fields) {
46
+ const volatileSet = new Set(volatile);
47
+ const reasons = [];
48
+ // Schema: does every declared field still exist, with the right type?
49
+ let schema = true;
50
+ const declared = new Set(fields ?? golden.items.flatMap((i) => Object.keys(i)));
51
+ for (const field of declared) {
52
+ const expected = golden.items.find((i) => i[field] !== null && i[field] !== undefined)?.[field];
53
+ const missing = actual.filter((i) => !(field in i) || i[field] === null || i[field] === undefined);
54
+ if (actual.length > 0 && missing.length === actual.length) {
55
+ schema = false;
56
+ reasons.push(`field "${field}" absent from every item`);
57
+ continue;
58
+ }
59
+ const mistyped = actual.filter((i) => i[field] != null && expected != null && typeof i[field] !== typeof expected);
60
+ if (mistyped.length > 0) {
61
+ schema = false;
62
+ reasons.push(`field "${field}" changed type in ${mistyped.length} items`);
63
+ }
64
+ }
65
+ // Semantic: the same items, as a set.
66
+ const wanted = golden.items.map((i) => key(i, volatileSet, fields));
67
+ const got = actual.map((i) => key(i, volatileSet, fields));
68
+ const pool = [...got];
69
+ const absent = [];
70
+ for (const k of wanted) {
71
+ const at = pool.indexOf(k);
72
+ if (at === -1)
73
+ absent.push(k);
74
+ else
75
+ pool.splice(at, 1);
76
+ }
77
+ const semantic = absent.length === 0 && pool.length === 0;
78
+ if (absent.length > 0)
79
+ reasons.push(`${absent.length} golden item(s) missing from the result`);
80
+ if (pool.length > 0)
81
+ reasons.push(`${pool.length} unexpected item(s) in the result`);
82
+ // Volatile fields must still be present and of the right type.
83
+ for (const field of volatileSet) {
84
+ const bad = actual.filter((i) => !(field in i) || i[field] === null || i[field] === undefined);
85
+ if (bad.length > 0) {
86
+ schema = false;
87
+ reasons.push(`volatile field "${field}" absent in ${bad.length} items`);
88
+ }
89
+ }
90
+ // Ordering: only answerable when the sets agree.
91
+ const ordering = semantic ? wanted.every((k, i) => got[i] === k) : null;
92
+ if (ordering === false)
93
+ reasons.push('same items, different order');
94
+ return { schema, semantic, ordering, reasons };
95
+ }
@@ -0,0 +1,146 @@
1
+ /** The fields the oracle compares; all of them when it names none. */
2
+ function comparedFields(items, oracle) {
3
+ return oracle.compare.fields ?? [...new Set(items.flatMap((i) => Object.keys(i)))];
4
+ }
5
+ const present = (value) => value !== null && value !== undefined && value !== '';
6
+ /**
7
+ * Layer 1. Decides whether the adjacent browser run is usable at all, before
8
+ * the engine is judged against it. Checks exactly the fields Layer 2 compares:
9
+ * disqualifying a baseline over a column nobody scores would throw away a
10
+ * usable measurement.
11
+ */
12
+ export function baselineValidity(items, oracle, opts) {
13
+ if (opts.threw)
14
+ return { usable: false, reason: 'baseline threw' };
15
+ if (items.length === 0) {
16
+ return opts.expectEmpty
17
+ ? { usable: true, reason: null }
18
+ : { usable: false, reason: 'baseline returned no items' };
19
+ }
20
+ for (const field of comparedFields(items, oracle)) {
21
+ const missing = items.filter((i) => !present(i[field])).length;
22
+ if (missing > items.length / 2) {
23
+ return { usable: false, reason: `baseline field "${field}" missing in ${missing}/${items.length} items` };
24
+ }
25
+ }
26
+ return { usable: true, reason: null };
27
+ }
28
+ function keyOf(item, fields) {
29
+ return JSON.stringify(fields.map((f) => [f, item[f] ?? null]));
30
+ }
31
+ /** Layer 2a. The same items, as a set, on the fields the oracle compares. */
32
+ export function entityEquivalence(reference, actual, oracle) {
33
+ const fields = comparedFields([...reference, ...actual], oracle);
34
+ const pool = actual.map((i) => keyOf(i, fields));
35
+ const reasons = [];
36
+ let missing = 0;
37
+ for (const key of reference.map((i) => keyOf(i, fields))) {
38
+ const at = pool.indexOf(key);
39
+ if (at === -1)
40
+ missing += 1;
41
+ else
42
+ pool.splice(at, 1);
43
+ }
44
+ if (missing > 0)
45
+ reasons.push(`${missing} baseline item(s) missing from the result`);
46
+ if (pool.length > 0)
47
+ reasons.push(`${pool.length} unexpected item(s) in the result`);
48
+ return { match: reasons.length === 0, reasons };
49
+ }
50
+ /**
51
+ * Layer 2b. The result answers the question that was asked.
52
+ *
53
+ * arbeitnow.com discards its search parameter on redirect and serves the same
54
+ * unfiltered front page for every query — a result that is internally
55
+ * consistent, matches its own baseline, and answers nothing. Entity equivalence
56
+ * alone cannot see that.
57
+ *
58
+ * Judged only where an oracle declares a rule, and returns null otherwise.
59
+ * Applying a generic "the query must appear in the results" heuristic
60
+ * everywhere would fail honest engines: a search for `senior rust engineer` may
61
+ * legitimately return `Systems Engineer`, a search for `cars` may return
62
+ * `automobile`, and a match may live in a description the oracle does not
63
+ * compare. A rule that fires on sites it was never designed for is an oracle
64
+ * bug wearing a correctness check's clothes.
65
+ */
66
+ export function querySemantics(actual, input, oracle) {
67
+ const rule = oracle.semantics;
68
+ if (!rule)
69
+ return { match: null, reasons: [] };
70
+ if (actual.length === 0)
71
+ return { match: null, reasons: [] };
72
+ const raw = input[rule.input];
73
+ if (raw === undefined || String(raw).trim() === '')
74
+ return { match: null, reasons: [] };
75
+ // Token-wise, so a multi-word query is not required to appear verbatim.
76
+ const tokens = String(raw).toLowerCase().split(/\s+/).filter((t) => t.length >= 2);
77
+ if (tokens.length === 0)
78
+ return { match: null, reasons: [] };
79
+ const hits = actual.filter((item) => rule.fields.some((f) => {
80
+ const value = String(item[f] ?? '').toLowerCase();
81
+ return tokens.some((t) => value.includes(t));
82
+ })).length;
83
+ return hits >= actual.length * rule.minShare
84
+ ? { match: true, reasons: [] }
85
+ : { match: false, reasons: [`no token of "${String(raw)}" in ${actual.length - hits}/${actual.length} items`] };
86
+ }
87
+ /** Layer 2c. Null when the oracle ignores order, or when the sets differ. */
88
+ export function orderingAgreement(reference, actual, oracle) {
89
+ if (oracle.compare.ordering === 'ignore')
90
+ return null;
91
+ if (!entityEquivalence(reference, actual, oracle).match)
92
+ return null;
93
+ const fields = comparedFields([...reference, ...actual], oracle);
94
+ return reference.every((item, i) => actual[i] !== undefined && keyOf(actual[i], fields) === keyOf(item, fields));
95
+ }
96
+ /** A result this much smaller than its golden suggests the plan, not the site. */
97
+ const COLLAPSE_RATIO = 0.5;
98
+ /**
99
+ * Compares a fresh browser result against its stored golden structurally rather
100
+ * than by content.
101
+ *
102
+ * Under `paired-live` the golden no longer scores correctness, but it still
103
+ * answers one question nothing else does: does the browser plan still see the
104
+ * page the way it used to? Items may legitimately all differ; the shape should
105
+ * not. Returns a description of the decay, or null.
106
+ */
107
+ export function planAging(golden, baseline, oracle) {
108
+ if (golden === undefined || golden.length === 0)
109
+ return null;
110
+ // A count change is a hint, not a verdict: a search that legitimately returns
111
+ // forty results today where it returned a hundred last week has not aged its
112
+ // plan. Field disappearance below is the far stronger signal.
113
+ if (baseline.length <= golden.length * COLLAPSE_RATIO) {
114
+ return `possible plan aging: item count changed ${golden.length} -> ${baseline.length}`;
115
+ }
116
+ for (const field of comparedFields(golden, oracle)) {
117
+ const hadIt = golden.filter((i) => present(i[field])).length > golden.length / 2;
118
+ const hasIt = baseline.filter((i) => present(i[field])).length > baseline.length / 2;
119
+ if (hadIt && !hasIt)
120
+ return `plan aging: field "${field}" no longer resolves`;
121
+ }
122
+ return null;
123
+ }
124
+ /** Engine equivalence, defined once so a report and a printed line cannot diverge. */
125
+ export const isEquivalent = (grade) => grade.entities === true && grade.query !== false;
126
+ /** An unusable baseline leaves layer two unjudged rather than failed. */
127
+ export function gradePair(reference, actual, oracle, opts) {
128
+ const validity = baselineValidity(reference, oracle, opts);
129
+ if (!validity.usable) {
130
+ return {
131
+ usable: false, validityReason: validity.reason, excludeReason: 'baseline',
132
+ entities: null, query: null, ordering: null, reasons: [validity.reason ?? 'baseline unusable'],
133
+ };
134
+ }
135
+ const entities = entityEquivalence(reference, actual, oracle);
136
+ const query = querySemantics(actual, opts.input, oracle);
137
+ return {
138
+ usable: true,
139
+ validityReason: null,
140
+ excludeReason: null,
141
+ entities: entities.match,
142
+ query: query.match,
143
+ ordering: orderingAgreement(reference, actual, oracle),
144
+ reasons: [...entities.reasons, ...query.reasons],
145
+ };
146
+ }
@@ -0,0 +1,35 @@
1
+ import { DATASET, PAGE_SIZE, search } from '../../fixtures/data.js';
2
+ const identity = (item) => String(item.url ?? '');
3
+ const KNOWN = new Set(DATASET.map((r) => `/item/${r.id}`));
4
+ const PAGES = Math.ceil(DATASET.length / PAGE_SIZE);
5
+ /** Every match, not just page one: a different ranking is not a wrong answer. */
6
+ export const datasetMatches = (query) => new Set(Array.from({ length: PAGES }, (_, i) => search(query, i + 1))
7
+ .flat()
8
+ .map((r) => `/item/${r.id}`));
9
+ /** Any page's worth of matches: a different ranking is not a wrong answer. */
10
+ export const DATASET_TRUTH = { expected: (input) => datasetMatches(String(input.query ?? '')) };
11
+ /** The one window this page should hold, which is what a page asks about. */
12
+ export const pageWindow = (input) => new Set(search(String(input.query ?? ''), Number(input.page ?? 1)).map((r) => `/item/${r.id}`));
13
+ export const PAGE_TRUTH = { expected: pageWindow };
14
+ /**
15
+ * Judges precision, not recall: every row returned must be one this query
16
+ * should have returned. Whether it returned all of them is a question about
17
+ * ranking and paging, which this benchmark does not ask and which a listing
18
+ * that inserts a fresh row would fail for no good reason.
19
+ */
20
+ export function judgeByGroundTruth(items, input, truth) {
21
+ if (items.length === 0)
22
+ return { match: null, reason: 'no items to judge' };
23
+ const expected = truth.expected(input);
24
+ for (const item of items) {
25
+ const id = identity(item);
26
+ if (expected.has(id))
27
+ continue;
28
+ if (truth.ephemeral?.test(id) === true)
29
+ continue;
30
+ return KNOWN.has(id)
31
+ ? { match: false, reason: `returned ${id}, a record this query does not match` }
32
+ : { match: false, reason: `returned ${id}, which is neither a known record nor a declared ephemeral row` };
33
+ }
34
+ return { match: true, reason: '' };
35
+ }