webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
import { mkdtemp, mkdir, writeFile, appendFile } from 'node:fs/promises';
|
|
2
|
+
import { tmpdir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { load } from 'cheerio';
|
|
5
|
+
import { teach } from '../authoring/teach.js';
|
|
6
|
+
import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
|
|
7
|
+
import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
|
|
8
|
+
import { PolitenessLayer } from '../net/politeness.js';
|
|
9
|
+
import { StaticSiteResolver, WILD_ORIGINS } from '../sites.js';
|
|
10
|
+
import { verifyReadable, UserError } from '../local.js';
|
|
11
|
+
import { PLANS } from '../../benchmark/plans.js';
|
|
12
|
+
/**
|
|
13
|
+
* What the frozen verifier says about real sites, beside what an independent
|
|
14
|
+
* account says the answer was.
|
|
15
|
+
*
|
|
16
|
+
* The verifier is not modified while this runs. A failure found on the first
|
|
17
|
+
* site and fixed before the last is a batch whose halves came from two
|
|
18
|
+
* different systems, and the first case is the one a fix overfits to.
|
|
19
|
+
*/
|
|
20
|
+
export const BASELINE = '7f51ca0';
|
|
21
|
+
/** At most four, plus every task that expects an empty result: those carry a
|
|
22
|
+
* deterministic oracle and are worth more than another ordinary query. */
|
|
23
|
+
const UNSEEN_PER_CASE = 4;
|
|
24
|
+
export function buildCases(tasks, domains) {
|
|
25
|
+
const groups = new Map();
|
|
26
|
+
for (const task of tasks) {
|
|
27
|
+
if (!domains.includes(task.site))
|
|
28
|
+
continue;
|
|
29
|
+
const key = `${task.site}\u0000${task.intent}`;
|
|
30
|
+
groups.set(key, [...(groups.get(key) ?? []), task]);
|
|
31
|
+
}
|
|
32
|
+
return [...groups.values()].map((all) => {
|
|
33
|
+
const [taught, ...rest] = all;
|
|
34
|
+
const empty = rest.filter((t) => t.expectEmpty === true);
|
|
35
|
+
const ordinary = rest.filter((t) => t.expectEmpty !== true).slice(0, UNSEEN_PER_CASE - empty.length);
|
|
36
|
+
return { site: taught.site, intent: taught.intent, taught: taught, unseen: [...ordinary, ...empty] };
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
/** Reads the signals each verifier already records; no verifier was changed. */
|
|
40
|
+
export function classifyNotTested(evidence) {
|
|
41
|
+
if (evidence === undefined)
|
|
42
|
+
return 'insufficient_evidence';
|
|
43
|
+
const declined = (evidence.probes ?? []).find((p) => p.unavailable !== undefined)?.unavailable ?? '';
|
|
44
|
+
if (/429|retry later/i.test(declined))
|
|
45
|
+
return 'rate_limited';
|
|
46
|
+
if (declined !== '')
|
|
47
|
+
return 'transport_failure';
|
|
48
|
+
const s = evidence.signals ?? {};
|
|
49
|
+
if (s.stable === false)
|
|
50
|
+
return 'volatile';
|
|
51
|
+
if (s.applicable === false)
|
|
52
|
+
return 'no_visible_lexical_match';
|
|
53
|
+
if (s.consistent === false)
|
|
54
|
+
return 'lexical_inconsistent';
|
|
55
|
+
if (s.agreed === false)
|
|
56
|
+
return 'browser_disagreement';
|
|
57
|
+
if (s.moved === false)
|
|
58
|
+
return 'page_control_ambiguous';
|
|
59
|
+
if ((evidence.probes ?? []).some((p) => p.role === 'alternate' && p.items === 0))
|
|
60
|
+
return 'empty_alternate_page';
|
|
61
|
+
if (!(evidence.probes ?? []).some((p) => p.role === 'contrast' || p.role === 'alternate'))
|
|
62
|
+
return 'insufficient_evidence';
|
|
63
|
+
return 'unclassified';
|
|
64
|
+
}
|
|
65
|
+
const WORD = /^[a-z][a-z0-9-]{3,}$/;
|
|
66
|
+
const tokens = (text) => [...new Set(text.toLowerCase().split(/[^a-z0-9-]+/i).filter((t) => WORD.test(t)))];
|
|
67
|
+
const carriesId = (items, id) => {
|
|
68
|
+
const wanted = id.toLowerCase();
|
|
69
|
+
return items.some((item) => Object.values(item).some((v) => String(v ?? '').toLowerCase().includes(wanted)));
|
|
70
|
+
};
|
|
71
|
+
/**
|
|
72
|
+
* Whether the id rule can be applied to this plan at all.
|
|
73
|
+
*
|
|
74
|
+
* Decided from the taught observation, which is the browser's reading of the
|
|
75
|
+
* url the task names and so is the right entity by the task's own definition.
|
|
76
|
+
* If the id cannot be seen even there, this plan does not extract anything that
|
|
77
|
+
* carries one and the rule tells us nothing about any other input.
|
|
78
|
+
*
|
|
79
|
+
* This answers only "can this oracle be used here". It is never the answer to
|
|
80
|
+
* "was this result right", which is what keeps it from grading itself.
|
|
81
|
+
*/
|
|
82
|
+
export function entityIdApplicable(taught, taughtItems) {
|
|
83
|
+
const type = 'entity_id';
|
|
84
|
+
if (taught.intent !== 'detail')
|
|
85
|
+
return { type, applicable: false, reason: 'not a detail lookup' };
|
|
86
|
+
const id = taught.input.id;
|
|
87
|
+
if (id === undefined || String(id) === '')
|
|
88
|
+
return { type, applicable: false, reason: 'the task names no entity id' };
|
|
89
|
+
if (taughtItems === null || taughtItems.length === 0)
|
|
90
|
+
return { type, applicable: false, reason: 'the taught observation could not be read' };
|
|
91
|
+
return carriesId(taughtItems, String(id))
|
|
92
|
+
? { type, applicable: true, reason: `the taught output carries the entity id ${id}` }
|
|
93
|
+
: { type, applicable: false, reason: 'requested entity id is not observable in the taught output' };
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* T1. A deterministic reading of the task itself, owing nothing to the probes.
|
|
97
|
+
*/
|
|
98
|
+
export function judgeByTask(task, items, role, entityId) {
|
|
99
|
+
if (task.expectEmpty === true) {
|
|
100
|
+
const oracle = { type: 'expect_empty', applicable: true, reason: 'the task declares this query matches nothing' };
|
|
101
|
+
return items.length === 0
|
|
102
|
+
? { judgement: 'correct', source: 'expect_empty', reason: 'a query declared to match nothing returned nothing', oracle }
|
|
103
|
+
: { judgement: 'wrong', source: 'expect_empty', reason: `a query declared to match nothing returned ${items.length} rows`, oracle };
|
|
104
|
+
}
|
|
105
|
+
if (!entityId.applicable)
|
|
106
|
+
return { judgement: 'indeterminate', source: 'none', reason: entityId.reason, oracle: entityId };
|
|
107
|
+
// The input the gate was calibrated on cannot also be judged by it.
|
|
108
|
+
if (role === 'taught') {
|
|
109
|
+
return {
|
|
110
|
+
judgement: 'indeterminate', source: 'none',
|
|
111
|
+
reason: 'this is the observation the id oracle was calibrated on, so it cannot also answer for it',
|
|
112
|
+
oracle: { ...entityId, applicable: false, reason: 'calibration input' },
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
const id = String(task.input.id);
|
|
116
|
+
if (items.length === 0) {
|
|
117
|
+
return { judgement: 'indeterminate', source: 'none', reason: 'nothing returned to check the id against', oracle: entityId };
|
|
118
|
+
}
|
|
119
|
+
return carriesId(items, id)
|
|
120
|
+
? { judgement: 'correct', source: 'entity_id', reason: `the requested id ${id} appears in the answer`, oracle: entityId }
|
|
121
|
+
: { judgement: 'wrong', source: 'entity_id', reason: `the requested id ${id} appears nowhere in the answer`, oracle: entityId };
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* T2. One-sided on purpose.
|
|
125
|
+
*
|
|
126
|
+
* A row that does not appear on the page it links to is a row describing
|
|
127
|
+
* something else, which is evidence the answer is wrong. A row that does appear
|
|
128
|
+
* there proves only that the row and its own link agree: move a whole row to
|
|
129
|
+
* the wrong entity — title, url and all — and this check still passes it. So a
|
|
130
|
+
* match settles nothing and goes to the next tier.
|
|
131
|
+
*/
|
|
132
|
+
export async function detailMismatch(items, net, origin, sample = 2) {
|
|
133
|
+
let requests = 0;
|
|
134
|
+
for (const item of items.slice(0, sample)) {
|
|
135
|
+
const href = String(item.url ?? '');
|
|
136
|
+
const title = String(item.title ?? '');
|
|
137
|
+
if (href === '' || title === '')
|
|
138
|
+
continue;
|
|
139
|
+
const wanted = tokens(title);
|
|
140
|
+
if (wanted.length === 0)
|
|
141
|
+
continue;
|
|
142
|
+
let url;
|
|
143
|
+
try {
|
|
144
|
+
url = new URL(href, origin).toString();
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
if (new URL(url).origin !== origin)
|
|
150
|
+
continue;
|
|
151
|
+
if (!(await net.isAllowed(url)))
|
|
152
|
+
continue;
|
|
153
|
+
let text;
|
|
154
|
+
try {
|
|
155
|
+
requests += 1;
|
|
156
|
+
const res = await net.fetch(url);
|
|
157
|
+
if (res.status !== 200)
|
|
158
|
+
continue;
|
|
159
|
+
text = load(res.body).text().toLowerCase();
|
|
160
|
+
}
|
|
161
|
+
catch {
|
|
162
|
+
continue;
|
|
163
|
+
}
|
|
164
|
+
// A shell that rendered nothing cannot disagree with anything.
|
|
165
|
+
if (text.length < 500)
|
|
166
|
+
continue;
|
|
167
|
+
if (!wanted.some((t) => text.includes(t))) {
|
|
168
|
+
return { wrong: true, requests, reason: `the row "${title}" shares no word with the page at ${href} that it links to` };
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
return { wrong: false, requests, reason: '' };
|
|
172
|
+
}
|
|
173
|
+
export const DISCOVERY_DOMAINS = [
|
|
174
|
+
'hex.pm', 'docs.rs', 'jsr.io', 'bandcamp.com',
|
|
175
|
+
'meta.discourse.org', 'lemmy.world', 'openlibrary.org', 'dev.to',
|
|
176
|
+
];
|
|
177
|
+
/** Untouched by this batch, and run once when a changed verifier is first evaluated. */
|
|
178
|
+
export const HELD_OUT_DOMAINS = ['npmjs.com', 'musicbrainz.org', 'mastodon.social', 'itch.io'];
|
|
179
|
+
export async function runDiscovery(cases, opts) {
|
|
180
|
+
await mkdir(join(opts.outDir, 'evidence'), { recursive: true });
|
|
181
|
+
const records = [];
|
|
182
|
+
const skips = [];
|
|
183
|
+
const net = new PolitenessLayer({ minIntervalMs: opts.minIntervalMs });
|
|
184
|
+
const skip = async (s) => {
|
|
185
|
+
skips.push(s);
|
|
186
|
+
opts.onSkip?.(s);
|
|
187
|
+
await appendFile(join(opts.outDir, 'skips.jsonl'), `${JSON.stringify(s)}\n`);
|
|
188
|
+
};
|
|
189
|
+
const keep = async (r) => {
|
|
190
|
+
records.push(r);
|
|
191
|
+
opts.onRecord?.(r);
|
|
192
|
+
await appendFile(join(opts.outDir, 'runs.jsonl'), `${JSON.stringify(r)}\n`);
|
|
193
|
+
};
|
|
194
|
+
for (const one of cases) {
|
|
195
|
+
const origin = WILD_ORIGINS[one.site];
|
|
196
|
+
const plan = PLANS[one.site]?.[one.intent];
|
|
197
|
+
if (origin === undefined || plan === undefined) {
|
|
198
|
+
await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'no_plan', detail: 'no hand-written plan for this site and intent' });
|
|
199
|
+
continue;
|
|
200
|
+
}
|
|
201
|
+
const taughtUrl = plan.url(origin, one.taught);
|
|
202
|
+
if (!(await net.isAllowed(taughtUrl))) {
|
|
203
|
+
await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'skipped_by_robots', detail: taughtUrl });
|
|
204
|
+
continue;
|
|
205
|
+
}
|
|
206
|
+
const planDir = await mkdtemp(join(tmpdir(), 'disc-plans-'));
|
|
207
|
+
const recipeDir = await mkdtemp(join(tmpdir(), 'disc-recipes-'));
|
|
208
|
+
const startedTeach = performance.now();
|
|
209
|
+
let taught;
|
|
210
|
+
try {
|
|
211
|
+
taught = await teach({
|
|
212
|
+
site: one.site, intent: one.intent, url: taughtUrl, input: one.taught.input,
|
|
213
|
+
itemSelector: plan.itemSelector, fields: plan.fields,
|
|
214
|
+
planDir, recipeDir, minIntervalMs: opts.minIntervalMs,
|
|
215
|
+
});
|
|
216
|
+
}
|
|
217
|
+
catch (error) {
|
|
218
|
+
await skip({ baseline: BASELINE, site: one.site, taskId: one.taught.id, reason: 'teach_failed', detail: error instanceof Error ? error.message : String(error) });
|
|
219
|
+
continue;
|
|
220
|
+
}
|
|
221
|
+
const teachMs = Math.round(performance.now() - startedTeach);
|
|
222
|
+
const verification = taught.plan.verification;
|
|
223
|
+
const contract = verification?.contract.required ?? [];
|
|
224
|
+
const notTestedReasons = {};
|
|
225
|
+
for (const check of contract) {
|
|
226
|
+
const evidence = verification?.evidence[check];
|
|
227
|
+
if (check === 'non_empty' || check === 'required_fields')
|
|
228
|
+
continue;
|
|
229
|
+
if (evidence?.status === 'passed' || evidence?.status === 'failed')
|
|
230
|
+
continue;
|
|
231
|
+
notTestedReasons[check] = classifyNotTested(evidence);
|
|
232
|
+
}
|
|
233
|
+
await writeFile(join(opts.outDir, 'evidence', `${one.site}-${one.intent}.json`), `${JSON.stringify({ baseline: BASELINE, site: one.site, intent: one.intent, taughtUrl, plan: taught.plan, refused: taught.refused, sample: taught.sample }, null, 2)}\n`);
|
|
234
|
+
const sites = new StaticSiteResolver({ [one.site]: origin });
|
|
235
|
+
const fields = Object.keys(plan.fields);
|
|
236
|
+
// Calibrated once per plan, from the browser's reading of the url the task
|
|
237
|
+
// names. It decides whether the id rule applies here, never what the answer was.
|
|
238
|
+
const entityId = entityIdApplicable(one.taught, taught.sample);
|
|
239
|
+
for (const [role, task] of [['taught', one.taught], ...one.unseen.map((t) => ['unseen', t])]) {
|
|
240
|
+
const url = plan.url(origin, task);
|
|
241
|
+
if (!(await net.isAllowed(url))) {
|
|
242
|
+
await skip({ baseline: BASELINE, site: one.site, taskId: task.id, reason: 'skipped_by_robots', detail: url });
|
|
243
|
+
continue;
|
|
244
|
+
}
|
|
245
|
+
const started = performance.now();
|
|
246
|
+
let items = null;
|
|
247
|
+
let error;
|
|
248
|
+
let requests = 0;
|
|
249
|
+
if (taught.recipe !== null) {
|
|
250
|
+
const strategy = taught.recipe.output.type === 'json' ? new HttpJsonStrategy(net, sites) : new HttpHtmlStrategy(net, sites);
|
|
251
|
+
try {
|
|
252
|
+
requests += 1;
|
|
253
|
+
const result = await strategy.execute(taught.recipe, { id: task.id, site: one.site, intent: one.intent, input: task.input });
|
|
254
|
+
items = result.status !== undefined && result.status !== taught.recipe.validation.status ? null : result.items;
|
|
255
|
+
if (items === null)
|
|
256
|
+
error = `status ${result.status}`;
|
|
257
|
+
}
|
|
258
|
+
catch (err) {
|
|
259
|
+
error = err instanceof Error ? err.message : String(err);
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
else {
|
|
263
|
+
error = `no recipe: ${taught.refused ?? 'unknown'}`;
|
|
264
|
+
}
|
|
265
|
+
let finalStatus = 'error';
|
|
266
|
+
let checks = {};
|
|
267
|
+
if (items !== null) {
|
|
268
|
+
try {
|
|
269
|
+
const read = verifyReadable(items, fields, verification);
|
|
270
|
+
finalStatus = read.status;
|
|
271
|
+
checks = read.checks;
|
|
272
|
+
}
|
|
273
|
+
catch (err) {
|
|
274
|
+
finalStatus = 'unverified';
|
|
275
|
+
if (err instanceof UserError)
|
|
276
|
+
error = err.message;
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
let { judgement, source, reason, oracle } = items === null
|
|
280
|
+
? {
|
|
281
|
+
judgement: 'indeterminate', source: 'none',
|
|
282
|
+
reason: error ?? 'no answer',
|
|
283
|
+
oracle: { type: 'none', applicable: false, reason: 'no answer to judge' },
|
|
284
|
+
}
|
|
285
|
+
: judgeByTask(task, items, role, entityId);
|
|
286
|
+
if (judgement === 'indeterminate' && items !== null && items.length > 0 && one.intent !== 'detail') {
|
|
287
|
+
const probe = await detailMismatch(items, net, origin);
|
|
288
|
+
requests += probe.requests;
|
|
289
|
+
if (probe.wrong) {
|
|
290
|
+
judgement = 'wrong';
|
|
291
|
+
source = 'detail_mismatch';
|
|
292
|
+
reason = probe.reason;
|
|
293
|
+
oracle = { type: 'detail_mismatch', applicable: true, reason: 'a returned row could be checked against the page it links to' };
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
// A person settles what reached verified, and a sample of what did not.
|
|
297
|
+
const needsManual = judgement === 'indeterminate' && (finalStatus === 'verified' || finalStatus === 'partially_verified');
|
|
298
|
+
await keep({
|
|
299
|
+
baseline: BASELINE, cached: false, site: one.site, taskId: task.id, intent: one.intent,
|
|
300
|
+
inputRole: role, input: task.input, contract, checks, finalStatus, notTestedReasons,
|
|
301
|
+
judgement, judgementSource: source, judgementReason: reason, oracle, needsManual,
|
|
302
|
+
items: items ?? [], itemCount: items?.length ?? 0,
|
|
303
|
+
cost: { httpRequests: requests, browserRuns: role === 'taught' ? 3 : 0, elapsedMs: Math.round(performance.now() - started) + (role === 'taught' ? teachMs : 0) },
|
|
304
|
+
...(error === undefined ? {} : { error }),
|
|
305
|
+
});
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
return { records, skips };
|
|
309
|
+
}
|
|
310
|
+
export function formatDiscovery(records, skips, eligible) {
|
|
311
|
+
const lines = [];
|
|
312
|
+
const judged = (j) => (r) => r.judgement === j;
|
|
313
|
+
const status = (s) => (r) => r.finalStatus === s;
|
|
314
|
+
const count = (f) => records.filter(f).length;
|
|
315
|
+
const both = (a, b) => count((r) => a(r) && b(r));
|
|
316
|
+
const domains = new Set(records.map((r) => r.site));
|
|
317
|
+
lines.push(`baseline: ${BASELINE} cached: false (every request hit the origin)`);
|
|
318
|
+
lines.push(`domains: ${domains.size} learned tasks: ${eligible} evaluated inputs: ${records.length}`);
|
|
319
|
+
const byReason = new Map();
|
|
320
|
+
for (const s of skips)
|
|
321
|
+
byReason.set(s.reason, (byReason.get(s.reason) ?? 0) + 1);
|
|
322
|
+
lines.push(`skipped: ${skips.length}${skips.length === 0 ? '' : ` (${[...byReason].map(([k, v]) => `${k} ${v}`).join(', ')})`}`);
|
|
323
|
+
lines.push('');
|
|
324
|
+
lines.push(' correct wrong indeterminate');
|
|
325
|
+
for (const s of ['verified', 'partially_verified', 'structural', 'unverified', 'error']) {
|
|
326
|
+
const row = records.filter(status(s));
|
|
327
|
+
if (row.length === 0)
|
|
328
|
+
continue;
|
|
329
|
+
lines.push(` ${s.padEnd(20)}${String(both(status(s), judged('correct'))).padStart(5)}${String(both(status(s), judged('wrong'))).padStart(8)}${String(both(status(s), judged('indeterminate'))).padStart(14)}`);
|
|
330
|
+
}
|
|
331
|
+
lines.push('');
|
|
332
|
+
const falseSuccess = records.filter((r) => r.finalStatus === 'verified' && r.judgement === 'wrong');
|
|
333
|
+
lines.push(`false success (verified + wrong): ${falseSuccess.length}`);
|
|
334
|
+
for (const r of falseSuccess)
|
|
335
|
+
lines.push(` ${r.site} ${r.taskId} ${JSON.stringify(r.input)} — ${r.judgementReason}`);
|
|
336
|
+
lines.push('');
|
|
337
|
+
lines.push('correct but not verified, by reason:');
|
|
338
|
+
const reasons = new Map();
|
|
339
|
+
for (const r of records) {
|
|
340
|
+
if (r.judgement !== 'correct' || r.finalStatus === 'verified')
|
|
341
|
+
continue;
|
|
342
|
+
for (const reason of Object.values(r.notTestedReasons))
|
|
343
|
+
reasons.set(reason, (reasons.get(reason) ?? 0) + 1);
|
|
344
|
+
if (Object.keys(r.notTestedReasons).length === 0)
|
|
345
|
+
reasons.set('structural_only', (reasons.get('structural_only') ?? 0) + 1);
|
|
346
|
+
}
|
|
347
|
+
for (const [reason, n] of [...reasons].sort((a, b) => b[1] - a[1]))
|
|
348
|
+
lines.push(` ${reason.padEnd(26)} ${n}`);
|
|
349
|
+
lines.push('');
|
|
350
|
+
const queue = records.filter((r) => r.needsManual);
|
|
351
|
+
lines.push(`awaiting manual judgement: ${queue.length}`);
|
|
352
|
+
for (const r of queue.slice(0, 40))
|
|
353
|
+
lines.push(` [${r.finalStatus}] ${r.site} ${r.taskId} ${JSON.stringify(r.input)} items=${r.itemCount}`);
|
|
354
|
+
return lines.join('\n');
|
|
355
|
+
}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
export function diffGolden(golden, actual, volatile = []) {
|
|
2
|
+
const reasons = [];
|
|
3
|
+
if (actual.length !== golden.items.length) {
|
|
4
|
+
reasons.push(`item count ${actual.length} !== golden ${golden.items.length}`);
|
|
5
|
+
return { match: false, reasons };
|
|
6
|
+
}
|
|
7
|
+
const volatileSet = new Set(volatile);
|
|
8
|
+
golden.items.forEach((expected, index) => {
|
|
9
|
+
const got = actual[index];
|
|
10
|
+
if (!got) {
|
|
11
|
+
reasons.push(`item ${index} missing`);
|
|
12
|
+
return;
|
|
13
|
+
}
|
|
14
|
+
for (const [field, expectedValue] of Object.entries(expected)) {
|
|
15
|
+
const gotValue = got[field];
|
|
16
|
+
if (volatileSet.has(field)) {
|
|
17
|
+
if (!(field in got) || gotValue === null || gotValue === undefined) {
|
|
18
|
+
reasons.push(`item ${index}: volatile field "${field}" absent`);
|
|
19
|
+
}
|
|
20
|
+
else if (typeof gotValue !== typeof expectedValue) {
|
|
21
|
+
reasons.push(`item ${index}: volatile field "${field}" changed type`);
|
|
22
|
+
}
|
|
23
|
+
continue;
|
|
24
|
+
}
|
|
25
|
+
if (gotValue !== expectedValue) {
|
|
26
|
+
reasons.push(`item ${index}: "${field}" ${JSON.stringify(gotValue)} !== ${JSON.stringify(expectedValue)}`);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
});
|
|
30
|
+
return { match: reasons.length === 0, reasons };
|
|
31
|
+
}
|
|
32
|
+
const key = (item, volatile, fields) => JSON.stringify((fields ?? Object.keys(item)).filter((k) => !volatile.has(k)).sort().map((k) => [k, item[k]]));
|
|
33
|
+
/**
|
|
34
|
+
* Three separate questions, because collapsing them mixes engine failure with
|
|
35
|
+
* website nondeterminism. flathub and pkg.go.dev reorder equally-ranked results
|
|
36
|
+
* between one request and the next, which failed even the browser baseline
|
|
37
|
+
* against its own golden — that is the site being unstable, not the recipe
|
|
38
|
+
* being wrong, and it should not be reported as the same thing.
|
|
39
|
+
*
|
|
40
|
+
* `fields`, when given, restricts both schema and semantic comparison to that
|
|
41
|
+
* set — the same restriction `oracle.compare.fields` applies under
|
|
42
|
+
* `paired-live`, so a column the oracle does not score cannot fail a golden
|
|
43
|
+
* task either.
|
|
44
|
+
*/
|
|
45
|
+
export function gradeGolden(golden, actual, volatile = [], fields) {
|
|
46
|
+
const volatileSet = new Set(volatile);
|
|
47
|
+
const reasons = [];
|
|
48
|
+
// Schema: does every declared field still exist, with the right type?
|
|
49
|
+
let schema = true;
|
|
50
|
+
const declared = new Set(fields ?? golden.items.flatMap((i) => Object.keys(i)));
|
|
51
|
+
for (const field of declared) {
|
|
52
|
+
const expected = golden.items.find((i) => i[field] !== null && i[field] !== undefined)?.[field];
|
|
53
|
+
const missing = actual.filter((i) => !(field in i) || i[field] === null || i[field] === undefined);
|
|
54
|
+
if (actual.length > 0 && missing.length === actual.length) {
|
|
55
|
+
schema = false;
|
|
56
|
+
reasons.push(`field "${field}" absent from every item`);
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
const mistyped = actual.filter((i) => i[field] != null && expected != null && typeof i[field] !== typeof expected);
|
|
60
|
+
if (mistyped.length > 0) {
|
|
61
|
+
schema = false;
|
|
62
|
+
reasons.push(`field "${field}" changed type in ${mistyped.length} items`);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
// Semantic: the same items, as a set.
|
|
66
|
+
const wanted = golden.items.map((i) => key(i, volatileSet, fields));
|
|
67
|
+
const got = actual.map((i) => key(i, volatileSet, fields));
|
|
68
|
+
const pool = [...got];
|
|
69
|
+
const absent = [];
|
|
70
|
+
for (const k of wanted) {
|
|
71
|
+
const at = pool.indexOf(k);
|
|
72
|
+
if (at === -1)
|
|
73
|
+
absent.push(k);
|
|
74
|
+
else
|
|
75
|
+
pool.splice(at, 1);
|
|
76
|
+
}
|
|
77
|
+
const semantic = absent.length === 0 && pool.length === 0;
|
|
78
|
+
if (absent.length > 0)
|
|
79
|
+
reasons.push(`${absent.length} golden item(s) missing from the result`);
|
|
80
|
+
if (pool.length > 0)
|
|
81
|
+
reasons.push(`${pool.length} unexpected item(s) in the result`);
|
|
82
|
+
// Volatile fields must still be present and of the right type.
|
|
83
|
+
for (const field of volatileSet) {
|
|
84
|
+
const bad = actual.filter((i) => !(field in i) || i[field] === null || i[field] === undefined);
|
|
85
|
+
if (bad.length > 0) {
|
|
86
|
+
schema = false;
|
|
87
|
+
reasons.push(`volatile field "${field}" absent in ${bad.length} items`);
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
// Ordering: only answerable when the sets agree.
|
|
91
|
+
const ordering = semantic ? wanted.every((k, i) => got[i] === k) : null;
|
|
92
|
+
if (ordering === false)
|
|
93
|
+
reasons.push('same items, different order');
|
|
94
|
+
return { schema, semantic, ordering, reasons };
|
|
95
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
/** The fields the oracle compares; all of them when it names none. */
|
|
2
|
+
function comparedFields(items, oracle) {
|
|
3
|
+
return oracle.compare.fields ?? [...new Set(items.flatMap((i) => Object.keys(i)))];
|
|
4
|
+
}
|
|
5
|
+
const present = (value) => value !== null && value !== undefined && value !== '';
|
|
6
|
+
/**
|
|
7
|
+
* Layer 1. Decides whether the adjacent browser run is usable at all, before
|
|
8
|
+
* the engine is judged against it. Checks exactly the fields Layer 2 compares:
|
|
9
|
+
* disqualifying a baseline over a column nobody scores would throw away a
|
|
10
|
+
* usable measurement.
|
|
11
|
+
*/
|
|
12
|
+
export function baselineValidity(items, oracle, opts) {
|
|
13
|
+
if (opts.threw)
|
|
14
|
+
return { usable: false, reason: 'baseline threw' };
|
|
15
|
+
if (items.length === 0) {
|
|
16
|
+
return opts.expectEmpty
|
|
17
|
+
? { usable: true, reason: null }
|
|
18
|
+
: { usable: false, reason: 'baseline returned no items' };
|
|
19
|
+
}
|
|
20
|
+
for (const field of comparedFields(items, oracle)) {
|
|
21
|
+
const missing = items.filter((i) => !present(i[field])).length;
|
|
22
|
+
if (missing > items.length / 2) {
|
|
23
|
+
return { usable: false, reason: `baseline field "${field}" missing in ${missing}/${items.length} items` };
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
return { usable: true, reason: null };
|
|
27
|
+
}
|
|
28
|
+
function keyOf(item, fields) {
|
|
29
|
+
return JSON.stringify(fields.map((f) => [f, item[f] ?? null]));
|
|
30
|
+
}
|
|
31
|
+
/** Layer 2a. The same items, as a set, on the fields the oracle compares. */
|
|
32
|
+
export function entityEquivalence(reference, actual, oracle) {
|
|
33
|
+
const fields = comparedFields([...reference, ...actual], oracle);
|
|
34
|
+
const pool = actual.map((i) => keyOf(i, fields));
|
|
35
|
+
const reasons = [];
|
|
36
|
+
let missing = 0;
|
|
37
|
+
for (const key of reference.map((i) => keyOf(i, fields))) {
|
|
38
|
+
const at = pool.indexOf(key);
|
|
39
|
+
if (at === -1)
|
|
40
|
+
missing += 1;
|
|
41
|
+
else
|
|
42
|
+
pool.splice(at, 1);
|
|
43
|
+
}
|
|
44
|
+
if (missing > 0)
|
|
45
|
+
reasons.push(`${missing} baseline item(s) missing from the result`);
|
|
46
|
+
if (pool.length > 0)
|
|
47
|
+
reasons.push(`${pool.length} unexpected item(s) in the result`);
|
|
48
|
+
return { match: reasons.length === 0, reasons };
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Layer 2b. The result answers the question that was asked.
|
|
52
|
+
*
|
|
53
|
+
* arbeitnow.com discards its search parameter on redirect and serves the same
|
|
54
|
+
* unfiltered front page for every query — a result that is internally
|
|
55
|
+
* consistent, matches its own baseline, and answers nothing. Entity equivalence
|
|
56
|
+
* alone cannot see that.
|
|
57
|
+
*
|
|
58
|
+
* Judged only where an oracle declares a rule, and returns null otherwise.
|
|
59
|
+
* Applying a generic "the query must appear in the results" heuristic
|
|
60
|
+
* everywhere would fail honest engines: a search for `senior rust engineer` may
|
|
61
|
+
* legitimately return `Systems Engineer`, a search for `cars` may return
|
|
62
|
+
* `automobile`, and a match may live in a description the oracle does not
|
|
63
|
+
* compare. A rule that fires on sites it was never designed for is an oracle
|
|
64
|
+
* bug wearing a correctness check's clothes.
|
|
65
|
+
*/
|
|
66
|
+
export function querySemantics(actual, input, oracle) {
|
|
67
|
+
const rule = oracle.semantics;
|
|
68
|
+
if (!rule)
|
|
69
|
+
return { match: null, reasons: [] };
|
|
70
|
+
if (actual.length === 0)
|
|
71
|
+
return { match: null, reasons: [] };
|
|
72
|
+
const raw = input[rule.input];
|
|
73
|
+
if (raw === undefined || String(raw).trim() === '')
|
|
74
|
+
return { match: null, reasons: [] };
|
|
75
|
+
// Token-wise, so a multi-word query is not required to appear verbatim.
|
|
76
|
+
const tokens = String(raw).toLowerCase().split(/\s+/).filter((t) => t.length >= 2);
|
|
77
|
+
if (tokens.length === 0)
|
|
78
|
+
return { match: null, reasons: [] };
|
|
79
|
+
const hits = actual.filter((item) => rule.fields.some((f) => {
|
|
80
|
+
const value = String(item[f] ?? '').toLowerCase();
|
|
81
|
+
return tokens.some((t) => value.includes(t));
|
|
82
|
+
})).length;
|
|
83
|
+
return hits >= actual.length * rule.minShare
|
|
84
|
+
? { match: true, reasons: [] }
|
|
85
|
+
: { match: false, reasons: [`no token of "${String(raw)}" in ${actual.length - hits}/${actual.length} items`] };
|
|
86
|
+
}
|
|
87
|
+
/** Layer 2c. Null when the oracle ignores order, or when the sets differ. */
|
|
88
|
+
export function orderingAgreement(reference, actual, oracle) {
|
|
89
|
+
if (oracle.compare.ordering === 'ignore')
|
|
90
|
+
return null;
|
|
91
|
+
if (!entityEquivalence(reference, actual, oracle).match)
|
|
92
|
+
return null;
|
|
93
|
+
const fields = comparedFields([...reference, ...actual], oracle);
|
|
94
|
+
return reference.every((item, i) => actual[i] !== undefined && keyOf(actual[i], fields) === keyOf(item, fields));
|
|
95
|
+
}
|
|
96
|
+
/** A result this much smaller than its golden suggests the plan, not the site. */
|
|
97
|
+
const COLLAPSE_RATIO = 0.5;
|
|
98
|
+
/**
|
|
99
|
+
* Compares a fresh browser result against its stored golden structurally rather
|
|
100
|
+
* than by content.
|
|
101
|
+
*
|
|
102
|
+
* Under `paired-live` the golden no longer scores correctness, but it still
|
|
103
|
+
* answers one question nothing else does: does the browser plan still see the
|
|
104
|
+
* page the way it used to? Items may legitimately all differ; the shape should
|
|
105
|
+
* not. Returns a description of the decay, or null.
|
|
106
|
+
*/
|
|
107
|
+
export function planAging(golden, baseline, oracle) {
|
|
108
|
+
if (golden === undefined || golden.length === 0)
|
|
109
|
+
return null;
|
|
110
|
+
// A count change is a hint, not a verdict: a search that legitimately returns
|
|
111
|
+
// forty results today where it returned a hundred last week has not aged its
|
|
112
|
+
// plan. Field disappearance below is the far stronger signal.
|
|
113
|
+
if (baseline.length <= golden.length * COLLAPSE_RATIO) {
|
|
114
|
+
return `possible plan aging: item count changed ${golden.length} -> ${baseline.length}`;
|
|
115
|
+
}
|
|
116
|
+
for (const field of comparedFields(golden, oracle)) {
|
|
117
|
+
const hadIt = golden.filter((i) => present(i[field])).length > golden.length / 2;
|
|
118
|
+
const hasIt = baseline.filter((i) => present(i[field])).length > baseline.length / 2;
|
|
119
|
+
if (hadIt && !hasIt)
|
|
120
|
+
return `plan aging: field "${field}" no longer resolves`;
|
|
121
|
+
}
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
/** Engine equivalence, defined once so a report and a printed line cannot diverge. */
|
|
125
|
+
export const isEquivalent = (grade) => grade.entities === true && grade.query !== false;
|
|
126
|
+
/** An unusable baseline leaves layer two unjudged rather than failed. */
|
|
127
|
+
export function gradePair(reference, actual, oracle, opts) {
|
|
128
|
+
const validity = baselineValidity(reference, oracle, opts);
|
|
129
|
+
if (!validity.usable) {
|
|
130
|
+
return {
|
|
131
|
+
usable: false, validityReason: validity.reason, excludeReason: 'baseline',
|
|
132
|
+
entities: null, query: null, ordering: null, reasons: [validity.reason ?? 'baseline unusable'],
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
const entities = entityEquivalence(reference, actual, oracle);
|
|
136
|
+
const query = querySemantics(actual, opts.input, oracle);
|
|
137
|
+
return {
|
|
138
|
+
usable: true,
|
|
139
|
+
validityReason: null,
|
|
140
|
+
excludeReason: null,
|
|
141
|
+
entities: entities.match,
|
|
142
|
+
query: query.match,
|
|
143
|
+
ordering: orderingAgreement(reference, actual, oracle),
|
|
144
|
+
reasons: [...entities.reasons, ...query.reasons],
|
|
145
|
+
};
|
|
146
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { DATASET, PAGE_SIZE, search } from '../../fixtures/data.js';
|
|
2
|
+
const identity = (item) => String(item.url ?? '');
|
|
3
|
+
const KNOWN = new Set(DATASET.map((r) => `/item/${r.id}`));
|
|
4
|
+
const PAGES = Math.ceil(DATASET.length / PAGE_SIZE);
|
|
5
|
+
/** Every match, not just page one: a different ranking is not a wrong answer. */
|
|
6
|
+
export const datasetMatches = (query) => new Set(Array.from({ length: PAGES }, (_, i) => search(query, i + 1))
|
|
7
|
+
.flat()
|
|
8
|
+
.map((r) => `/item/${r.id}`));
|
|
9
|
+
/** Any page's worth of matches: a different ranking is not a wrong answer. */
|
|
10
|
+
export const DATASET_TRUTH = { expected: (input) => datasetMatches(String(input.query ?? '')) };
|
|
11
|
+
/** The one window this page should hold, which is what a page asks about. */
|
|
12
|
+
export const pageWindow = (input) => new Set(search(String(input.query ?? ''), Number(input.page ?? 1)).map((r) => `/item/${r.id}`));
|
|
13
|
+
export const PAGE_TRUTH = { expected: pageWindow };
|
|
14
|
+
/**
|
|
15
|
+
* Judges precision, not recall: every row returned must be one this query
|
|
16
|
+
* should have returned. Whether it returned all of them is a question about
|
|
17
|
+
* ranking and paging, which this benchmark does not ask and which a listing
|
|
18
|
+
* that inserts a fresh row would fail for no good reason.
|
|
19
|
+
*/
|
|
20
|
+
export function judgeByGroundTruth(items, input, truth) {
|
|
21
|
+
if (items.length === 0)
|
|
22
|
+
return { match: null, reason: 'no items to judge' };
|
|
23
|
+
const expected = truth.expected(input);
|
|
24
|
+
for (const item of items) {
|
|
25
|
+
const id = identity(item);
|
|
26
|
+
if (expected.has(id))
|
|
27
|
+
continue;
|
|
28
|
+
if (truth.ephemeral?.test(id) === true)
|
|
29
|
+
continue;
|
|
30
|
+
return KNOWN.has(id)
|
|
31
|
+
? { match: false, reason: `returned ${id}, a record this query does not match` }
|
|
32
|
+
: { match: false, reason: `returned ${id}, which is neither a known record nor a declared ephemeral row` };
|
|
33
|
+
}
|
|
34
|
+
return { match: true, reason: '' };
|
|
35
|
+
}
|