webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
import { SETTLE_MS } from '../browser/navigate.js';
|
|
2
|
+
import { openSession } from '../browser/session.js';
|
|
3
|
+
import { DEFAULT_USER_AGENT, PolitenessLayer } from '../net/politeness.js';
|
|
4
|
+
/**
|
|
5
|
+
* Held-out 4's accessibility screen was robots.txt plus one GET, and it passed
|
|
6
|
+
* three sites that turned out to be boundaries: a virtualised timeline the plan
|
|
7
|
+
* sees one row of, a page that does not render in time, and a site that hands
|
|
8
|
+
* the engine a 200 challenge. Each of those is visible before a plan exists,
|
|
9
|
+
* but only by putting the engine's client and a real browser side by side.
|
|
10
|
+
*
|
|
11
|
+
* This measures; it does not judge. Even the line about the two pages differing
|
|
12
|
+
* names the evidence rather than returning a verdict: mastodon.social reads as
|
|
13
|
+
* a title mismatch because its SPA shell is titled differently, which is true
|
|
14
|
+
* and is not musicbrainz's kind of discrimination. The human decides which it is.
|
|
15
|
+
*/
|
|
16
|
+
const NAVIGATION_TIMEOUT_MS = 30_000;
|
|
17
|
+
/** A body this much smaller than the browser's is not the same page. */
|
|
18
|
+
const SIZE_RATIO = 0.1;
|
|
19
|
+
function titleOf(html) {
|
|
20
|
+
return /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html)?.[1]?.trim() ?? null;
|
|
21
|
+
}
|
|
22
|
+
/** The longest array anywhere in a payload, which is what a listing arrives as. */
|
|
23
|
+
export function largestArray(value) {
|
|
24
|
+
if (Array.isArray(value)) {
|
|
25
|
+
return value.reduce((best, v) => Math.max(best, largestArray(v)), value.length);
|
|
26
|
+
}
|
|
27
|
+
if (value !== null && typeof value === 'object') {
|
|
28
|
+
return Object.values(value).reduce((best, v) => Math.max(best, largestArray(v)), 0);
|
|
29
|
+
}
|
|
30
|
+
return 0;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* The largest run of sibling elements sharing a tag+class signature, which is
|
|
34
|
+
* the same heuristic the authoring scratch scripts used to find a listing
|
|
35
|
+
* before anyone had written a selector for it.
|
|
36
|
+
*/
|
|
37
|
+
const LARGEST_SIBLING_RUN = `(() => {
|
|
38
|
+
let best = { count: 0, signature: '' }
|
|
39
|
+
const walk = (parent) => {
|
|
40
|
+
const counts = new Map()
|
|
41
|
+
for (const child of parent.children) {
|
|
42
|
+
const cls = (child.getAttribute('class') || '').trim().split(/\\s+/).filter(Boolean).join('.')
|
|
43
|
+
const signature = child.tagName.toLowerCase() + (cls ? '.' + cls : '')
|
|
44
|
+
counts.set(signature, (counts.get(signature) || 0) + 1)
|
|
45
|
+
}
|
|
46
|
+
for (const [signature, count] of counts) {
|
|
47
|
+
if (count > best.count) best = { count, signature }
|
|
48
|
+
}
|
|
49
|
+
for (const child of parent.children) walk(child)
|
|
50
|
+
}
|
|
51
|
+
if (document.body) walk(document.body)
|
|
52
|
+
return best.count > 0 ? best : null
|
|
53
|
+
})()`;
|
|
54
|
+
export async function screen(url, net) {
|
|
55
|
+
const client = net ?? new PolitenessLayer({ userAgent: DEFAULT_USER_AGENT });
|
|
56
|
+
const fetched = await client.fetch(url);
|
|
57
|
+
const session = await openSession();
|
|
58
|
+
let largestJsonArray = 0;
|
|
59
|
+
const bodies = [];
|
|
60
|
+
session.page.on('response', (response) => {
|
|
61
|
+
if (!/json/i.test(response.headers()['content-type'] ?? ''))
|
|
62
|
+
return;
|
|
63
|
+
bodies.push(response
|
|
64
|
+
.text()
|
|
65
|
+
.then((text) => { largestJsonArray = Math.max(largestJsonArray, largestArray(JSON.parse(text))); })
|
|
66
|
+
.catch(() => undefined));
|
|
67
|
+
});
|
|
68
|
+
try {
|
|
69
|
+
const started = Date.now();
|
|
70
|
+
const response = await session.page.goto(url, {
|
|
71
|
+
waitUntil: 'domcontentloaded',
|
|
72
|
+
timeout: NAVIGATION_TIMEOUT_MS,
|
|
73
|
+
});
|
|
74
|
+
// Idle is the condition a slow page fails, so it gets the full navigation
|
|
75
|
+
// budget rather than navigate.ts's short best-effort wait. A run that ends
|
|
76
|
+
// at the timeout is the measurement, which is why whether it arrived is
|
|
77
|
+
// reported beside the number.
|
|
78
|
+
const networkIdle = await session.page
|
|
79
|
+
.waitForLoadState('networkidle', { timeout: NAVIGATION_TIMEOUT_MS })
|
|
80
|
+
.then(() => true)
|
|
81
|
+
.catch(() => false);
|
|
82
|
+
await session.page.waitForTimeout(SETTLE_MS);
|
|
83
|
+
const renderMs = Date.now() - started;
|
|
84
|
+
// A body still parsing is part of the measurement. Left to race the report,
|
|
85
|
+
// it is silently missed, and a missed array is what a virtualised list
|
|
86
|
+
// looks like.
|
|
87
|
+
await Promise.all(bodies);
|
|
88
|
+
const html = await session.page.content();
|
|
89
|
+
const browser = {
|
|
90
|
+
status: response?.status() ?? 0,
|
|
91
|
+
bytes: Buffer.byteLength(html, 'utf8'),
|
|
92
|
+
title: titleOf(html),
|
|
93
|
+
};
|
|
94
|
+
const engine = {
|
|
95
|
+
status: fetched.status,
|
|
96
|
+
bytes: fetched.bytesDownloaded,
|
|
97
|
+
title: titleOf(fetched.body),
|
|
98
|
+
};
|
|
99
|
+
const carriesBrowserTitle = browser.title !== null && browser.title !== '' && fetched.body.includes(browser.title);
|
|
100
|
+
return {
|
|
101
|
+
url,
|
|
102
|
+
engine,
|
|
103
|
+
browser,
|
|
104
|
+
carriesBrowserTitle,
|
|
105
|
+
differences: differencesBetween(engine, browser),
|
|
106
|
+
renderMs,
|
|
107
|
+
networkIdle,
|
|
108
|
+
largestListing: (await session.page.evaluate(LARGEST_SIBLING_RUN)),
|
|
109
|
+
largestJsonArray,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
finally {
|
|
113
|
+
await session.close();
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
function differencesBetween(engine, browser) {
|
|
117
|
+
const differences = [];
|
|
118
|
+
if (engine.title !== browser.title)
|
|
119
|
+
differences.push('title mismatch');
|
|
120
|
+
if (browser.bytes > 0 && engine.bytes < browser.bytes * SIZE_RATIO) {
|
|
121
|
+
differences.push(`engine body is ${Math.round((engine.bytes / browser.bytes) * 100)}% of browser`);
|
|
122
|
+
}
|
|
123
|
+
return differences;
|
|
124
|
+
}
|
|
125
|
+
function kb(bytes) {
|
|
126
|
+
return `${(bytes / 1024).toFixed(1)}KB`;
|
|
127
|
+
}
|
|
128
|
+
function side(name, s) {
|
|
129
|
+
return ` ${name.padEnd(8)} ${String(s.status).padEnd(4)} ${kb(s.bytes).padStart(8)} ${JSON.stringify(s.title)}`;
|
|
130
|
+
}
|
|
131
|
+
export function formatScreen(r) {
|
|
132
|
+
const listing = r.largestListing === null
|
|
133
|
+
? 'largest listing: none'
|
|
134
|
+
: `largest listing: ${r.largestListing.count} × ${r.largestListing.signature}`;
|
|
135
|
+
return [
|
|
136
|
+
r.url,
|
|
137
|
+
side('engine', r.engine),
|
|
138
|
+
side('browser', r.browser),
|
|
139
|
+
` engine body carries the browser title: ${r.carriesBrowserTitle ? 'yes' : 'no'}`,
|
|
140
|
+
` engine page differs from browser page: ${r.differences.join(', ') || 'no'}`,
|
|
141
|
+
` render: ${r.renderMs}ms${r.networkIdle ? '' : ' (networkidle not reached)'}`,
|
|
142
|
+
` ${listing} largest json array: ${r.largestJsonArray}`,
|
|
143
|
+
].join('\n');
|
|
144
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { elementsOf, generateCandidates, loadPage, normalize, perceivedText } from '../authoring/candidates.js';
|
|
2
|
+
/** `/item/1` must not answer for `/item/12`, so the match has to end at a boundary. */
|
|
3
|
+
function hrefCovers(href, item) {
|
|
4
|
+
let from = 0;
|
|
5
|
+
for (;;) {
|
|
6
|
+
const at = href.indexOf(item, from);
|
|
7
|
+
if (at === -1)
|
|
8
|
+
return false;
|
|
9
|
+
const after = href[at + item.length];
|
|
10
|
+
if (after === undefined || !/[A-Za-z0-9_-]/.test(after))
|
|
11
|
+
return true;
|
|
12
|
+
from = at + 1;
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
function coversOf($, el, label) {
|
|
16
|
+
if (label.identifier === 'text') {
|
|
17
|
+
const text = perceivedText($, el);
|
|
18
|
+
return label.items.flatMap((item, i) => (text.includes(normalize(item)) ? [i] : []));
|
|
19
|
+
}
|
|
20
|
+
const own = $(el).attr('href');
|
|
21
|
+
const hrefs = [...(own === undefined ? [] : [own]), ...$(el).find('[href]').map((_, n) => $(n).attr('href') ?? '').toArray()];
|
|
22
|
+
return label.items.flatMap((item, i) => (hrefs.some((href) => hrefCovers(href, item)) ? [i] : []));
|
|
23
|
+
}
|
|
24
|
+
function scoreAgainst($, selector, label) {
|
|
25
|
+
const matched = elementsOf($, selector);
|
|
26
|
+
if (matched.length === 0)
|
|
27
|
+
return { precision: 0, recall: 0, exact: false };
|
|
28
|
+
const covers = matched.map((el) => coversOf($, el, label));
|
|
29
|
+
const covered = new Set(covers.flat());
|
|
30
|
+
const precision = covers.filter((c) => c.length > 0).length / matched.length;
|
|
31
|
+
const recall = covered.size / label.items.length;
|
|
32
|
+
// A single <ul> wrapping every item scores 1.00 on both ratios and is not an
|
|
33
|
+
// item selector, so exactness asks for a one-to-one correspondence as well.
|
|
34
|
+
const onePerMatch = covers.every((c) => c.length === 1);
|
|
35
|
+
const onePerItem = label.items.every((_, i) => covers.filter((c) => c.includes(i)).length === 1);
|
|
36
|
+
const exact = matched.length === label.items.length && onePerMatch && onePerItem;
|
|
37
|
+
return { precision, recall, exact };
|
|
38
|
+
}
|
|
39
|
+
export function scoreSelector(html, selector, label) {
|
|
40
|
+
return scoreAgainst(loadPage(html), selector, label);
|
|
41
|
+
}
|
|
42
|
+
export function scoreSnapshot(snapshot, html, label) {
|
|
43
|
+
const $ = loadPage(html);
|
|
44
|
+
const rows = generateCandidates(html).map((candidate) => ({
|
|
45
|
+
candidate,
|
|
46
|
+
score: scoreAgainst($, candidate.selector, label),
|
|
47
|
+
}));
|
|
48
|
+
const found = rows.findIndex((r) => r.score.exact);
|
|
49
|
+
return {
|
|
50
|
+
snapshot,
|
|
51
|
+
labeled: label.items.length,
|
|
52
|
+
rows,
|
|
53
|
+
exactRank: found === -1 ? null : found + 1,
|
|
54
|
+
topIsExact: rows[0]?.score.exact === true,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
const pct = (value) => value.toFixed(2);
|
|
58
|
+
/** Rows are for a human to scan; the measurement itself runs over the full list. */
|
|
59
|
+
const MAX_ROWS_SHOWN = 50;
|
|
60
|
+
function formatSnapshotReport(report) {
|
|
61
|
+
const lines = [
|
|
62
|
+
`${report.snapshot} (${report.labeled} labeled items)`,
|
|
63
|
+
` generator: an exact candidate is present ${report.exactRank === null ? 'no' : `yes (rank ${report.exactRank} of ${report.rows.length})`}`,
|
|
64
|
+
` heuristic: top candidate is exact ${report.topIsExact ? 'yes' : 'no'}`,
|
|
65
|
+
];
|
|
66
|
+
report.rows.slice(0, MAX_ROWS_SHOWN).forEach(({ candidate, score }, i) => {
|
|
67
|
+
const mark = score.exact ? ' <- exact' : '';
|
|
68
|
+
const scope = candidate.scoped ? '' : ' (unscoped)';
|
|
69
|
+
lines.push(` ${String(i + 1).padStart(2)}. ${candidate.selector.padEnd(44)} ${String(candidate.count).padStart(4)}` +
|
|
70
|
+
` p=${pct(score.precision)} r=${pct(score.recall)}${mark}${scope}`);
|
|
71
|
+
});
|
|
72
|
+
if (report.rows.length > MAX_ROWS_SHOWN) {
|
|
73
|
+
lines.push(` ... and ${report.rows.length - MAX_ROWS_SHOWN} further candidates, not shown`);
|
|
74
|
+
}
|
|
75
|
+
return lines.join('\n');
|
|
76
|
+
}
|
|
77
|
+
export function formatCorpus(reports) {
|
|
78
|
+
const generated = reports.filter((r) => r.exactRank !== null).length;
|
|
79
|
+
const ranked = reports.filter((r) => r.topIsExact).length;
|
|
80
|
+
return [
|
|
81
|
+
...reports.map(formatSnapshotReport),
|
|
82
|
+
'',
|
|
83
|
+
`generator recall: ${generated}/${reports.length}`,
|
|
84
|
+
`ranking accuracy: ${ranked}/${reports.length}`,
|
|
85
|
+
].join('\n\n');
|
|
86
|
+
}
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
import { mkdtemp } from 'node:fs/promises';
|
|
2
|
+
import { tmpdir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { teach } from '../authoring/teach.js';
|
|
5
|
+
import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
|
|
6
|
+
import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
|
|
7
|
+
import { PolitenessLayer } from '../net/politeness.js';
|
|
8
|
+
import { StaticSiteResolver } from '../sites.js';
|
|
9
|
+
import { judgeByGroundTruth, DATASET_TRUTH, PAGE_TRUTH, pageWindow } from './ground-truth.js';
|
|
10
|
+
import { structurallySuccessful } from './verification-matrix.js';
|
|
11
|
+
import { CheckNameSchema } from '../local.js';
|
|
12
|
+
import { startSsrFixture } from '../../fixtures/ssr.js';
|
|
13
|
+
import { startIgnoringFixture } from '../../fixtures/ignoring.js';
|
|
14
|
+
import { startShiftedFixture } from '../../fixtures/shifted.js';
|
|
15
|
+
import { startVolatileFixture } from '../../fixtures/volatile.js';
|
|
16
|
+
import { startLimitingFixture } from '../../fixtures/limiting.js';
|
|
17
|
+
import { startCloakingFixture } from '../../fixtures/cloaking.js';
|
|
18
|
+
import { startPageIgnoringFixture, startPinnedFixture, startPageLimitedFixture } from '../../fixtures/paging.js';
|
|
19
|
+
import { search } from '../../fixtures/data.js';
|
|
20
|
+
/**
|
|
21
|
+
* The first set: one fixture per behaviour the verifier has to tell apart.
|
|
22
|
+
*
|
|
23
|
+
* Small on purpose. What is being measured is whether the plumbing can show a
|
|
24
|
+
* false success at all, not how a large site population behaves.
|
|
25
|
+
*/
|
|
26
|
+
export const VERIFICATION_CASES = [
|
|
27
|
+
{ id: 'honest', start: startSsrFixture, query: 'rust', truth: DATASET_TRUTH },
|
|
28
|
+
{ id: 'ignoring', start: startIgnoringFixture, query: 'rust', truth: DATASET_TRUTH },
|
|
29
|
+
{ id: 'shifted', start: startShiftedFixture, query: 'rust', truth: DATASET_TRUTH },
|
|
30
|
+
// The one fixture that invents a row, and the only one allowed to.
|
|
31
|
+
{ id: 'volatile', start: startVolatileFixture, query: 'rust',
|
|
32
|
+
truth: { expected: DATASET_TRUTH.expected, ephemeral: /^\/item\/live-\d+$/ } },
|
|
33
|
+
{ id: 'rate-limited', start: () => startLimitingFixture(429), query: 'rust', truth: DATASET_TRUTH },
|
|
34
|
+
{ id: 'cloaking', start: startCloakingFixture, query: 'rust', truth: DATASET_TRUTH },
|
|
35
|
+
];
|
|
36
|
+
const FIELDS = { title: 'a.title', url: 'a.title@href' };
|
|
37
|
+
/**
|
|
38
|
+
* Pagination cases, counted apart from the query ones.
|
|
39
|
+
*
|
|
40
|
+
* A separate denominator on purpose: these fixtures are built to exercise a
|
|
41
|
+
* page control and the query ones are not, so pooling them would produce a rate
|
|
42
|
+
* over a population nobody chose.
|
|
43
|
+
*/
|
|
44
|
+
export const PAGINATION_CASES = [
|
|
45
|
+
{ id: 'paged', start: startSsrFixture, query: 'senior', page: 1,
|
|
46
|
+
answerInput: { query: 'senior', page: 2 }, truth: PAGE_TRUTH },
|
|
47
|
+
{ id: 'page-ignored', start: () => startPageIgnoringFixture({ browserPaginates: false }), query: 'senior', page: 1,
|
|
48
|
+
answerInput: { query: 'senior', page: 2 }, truth: PAGE_TRUTH },
|
|
49
|
+
{ id: 'page-stale', start: () => startPageIgnoringFixture({ browserPaginates: true }), query: 'senior', page: 1,
|
|
50
|
+
answerInput: { query: 'senior', page: 2 }, truth: PAGE_TRUTH },
|
|
51
|
+
// Eight matches, so page two is genuinely empty and page one is the answer.
|
|
52
|
+
{ id: 'single-page', start: startSsrFixture, query: 'elixir', page: 1,
|
|
53
|
+
answerInput: { query: 'elixir', page: 1 }, truth: PAGE_TRUTH },
|
|
54
|
+
// Its pinned row belongs to page one and is declared, not tolerated by a
|
|
55
|
+
// general rule that would excuse any unrecognised row.
|
|
56
|
+
{ id: 'page-pinned', start: startPinnedFixture, query: 'senior', page: 1,
|
|
57
|
+
answerInput: { query: 'senior', page: 2 },
|
|
58
|
+
truth: { expected: (input) => new Set([...pageWindow(input), `/item/${search(String(input.query ?? ''), 1)[0].id}`]) } },
|
|
59
|
+
{ id: 'page-volatile', start: startVolatileFixture, query: 'senior', page: 1,
|
|
60
|
+
answerInput: { query: 'senior', page: 2 },
|
|
61
|
+
truth: { expected: pageWindow, ephemeral: /^\/item\/live-\d+$/ } },
|
|
62
|
+
// Answers page one and declines the rest, so the probe is declined while the
|
|
63
|
+
// answer being judged is fine.
|
|
64
|
+
{ id: 'page-limited', start: () => startPageLimitedFixture(503, 1), query: 'senior', page: 1,
|
|
65
|
+
answerInput: { query: 'senior', page: 1 }, truth: PAGE_TRUTH },
|
|
66
|
+
{ id: 'page-cloaking', start: startCloakingFixture, query: 'senior', page: 1,
|
|
67
|
+
answerInput: { query: 'senior', page: 2 }, truth: PAGE_TRUTH },
|
|
68
|
+
];
|
|
69
|
+
export async function runCase(one, opts = {}) {
|
|
70
|
+
const fixture = await one.start();
|
|
71
|
+
try {
|
|
72
|
+
const planDir = await mkdtemp(join(tmpdir(), 'matrix-plans-'));
|
|
73
|
+
const recipeDir = await mkdtemp(join(tmpdir(), 'matrix-recipes-'));
|
|
74
|
+
const site = `case-${one.id}`;
|
|
75
|
+
const input = one.page === undefined
|
|
76
|
+
? { query: one.query }
|
|
77
|
+
: { query: one.query, page: one.page };
|
|
78
|
+
const url = one.page === undefined
|
|
79
|
+
? `${fixture.url}/search?q=${encodeURIComponent(one.query)}`
|
|
80
|
+
: `${fixture.url}/search?q=${encodeURIComponent(one.query)}&page=${one.page}`;
|
|
81
|
+
const taught = await teach({
|
|
82
|
+
site, intent: 'search',
|
|
83
|
+
url,
|
|
84
|
+
input,
|
|
85
|
+
itemSelector: 'li.result', fields: FIELDS,
|
|
86
|
+
planDir, recipeDir, minIntervalMs: 0,
|
|
87
|
+
...(opts.skipSemanticVerification === true ? { skipSemanticVerification: true } : {}),
|
|
88
|
+
});
|
|
89
|
+
const evidence = taught.plan.verification?.evidence ?? {};
|
|
90
|
+
const checks = {};
|
|
91
|
+
const reasons = {};
|
|
92
|
+
for (const name of CheckNameSchema.options) {
|
|
93
|
+
const entry = evidence[name];
|
|
94
|
+
if (entry === undefined || name === 'non_empty' || name === 'required_fields')
|
|
95
|
+
continue;
|
|
96
|
+
checks[name] = entry.status;
|
|
97
|
+
if (entry.reason !== undefined)
|
|
98
|
+
reasons[name] = entry.reason;
|
|
99
|
+
}
|
|
100
|
+
// The answer a run would actually return: the recipe, not the recording.
|
|
101
|
+
const answerInput = one.answerInput ?? input;
|
|
102
|
+
const items = taught.recipe === null ? null : await answerOf(taught.recipe, site, fixture.url, answerInput);
|
|
103
|
+
const oracle = items === null
|
|
104
|
+
? { match: null, reason: 'no answer to judge' }
|
|
105
|
+
: judgeByGroundTruth(items, answerInput, one.truth);
|
|
106
|
+
return {
|
|
107
|
+
id: one.id,
|
|
108
|
+
checks,
|
|
109
|
+
oracleMatch: oracle.match,
|
|
110
|
+
structuralSuccess: items !== null && structurallySuccessful(items, Object.keys(FIELDS)),
|
|
111
|
+
...(oracle.reason === '' ? {} : { oracleReason: oracle.reason }),
|
|
112
|
+
...(Object.keys(reasons).length === 0 ? {} : { reasons }),
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
finally {
|
|
116
|
+
await fixture.close();
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
async function answerOf(recipe, site, origin, input) {
|
|
120
|
+
const net = new PolitenessLayer({ minIntervalMs: 0 });
|
|
121
|
+
const sites = new StaticSiteResolver({ [site]: origin });
|
|
122
|
+
const strategy = recipe.output.type === 'json' ? new HttpJsonStrategy(net, sites) : new HttpHtmlStrategy(net, sites);
|
|
123
|
+
try {
|
|
124
|
+
const result = await strategy.execute(recipe, { id: 'matrix', site, intent: 'search', input });
|
|
125
|
+
return result.status !== undefined && result.status !== recipe.validation.status ? null : result.items;
|
|
126
|
+
}
|
|
127
|
+
catch {
|
|
128
|
+
return null;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
export async function runVerificationCases(cases = VERIFICATION_CASES, opts = {}) {
|
|
132
|
+
const outcomes = [];
|
|
133
|
+
// Sequential: each case launches a browser, and running them at once would
|
|
134
|
+
// measure contention rather than the sites.
|
|
135
|
+
for (const one of cases)
|
|
136
|
+
outcomes.push(await runCase(one, opts));
|
|
137
|
+
return outcomes;
|
|
138
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a contract requiring exactly these checks would conclude.
|
|
3
|
+
*
|
|
4
|
+
* Scoring a layer means asking this with a shorter list, which is how adding a
|
|
5
|
+
* check can be read as a change rather than as a new number with no ancestor.
|
|
6
|
+
*/
|
|
7
|
+
export function verdictFrom(outcome, required) {
|
|
8
|
+
const verdicts = required.map((check) => outcome.checks[check] ?? 'not_tested');
|
|
9
|
+
if (verdicts.includes('failed'))
|
|
10
|
+
return 'failed';
|
|
11
|
+
return verdicts.every((v) => v === 'passed') ? 'passed' : 'not_tested';
|
|
12
|
+
}
|
|
13
|
+
export function tabulate(outcomes, required) {
|
|
14
|
+
const verdict = (o) => verdictFrom(o, required);
|
|
15
|
+
const judged = outcomes.filter((o) => o.oracleMatch !== null);
|
|
16
|
+
const passed = outcomes.filter((o) => verdict(o) === 'passed');
|
|
17
|
+
const withheld = judged.filter((o) => verdict(o) !== 'passed');
|
|
18
|
+
return {
|
|
19
|
+
cases: outcomes.length,
|
|
20
|
+
judged: judged.length,
|
|
21
|
+
passed: passed.length,
|
|
22
|
+
failed: outcomes.filter((o) => verdict(o) === 'failed').length,
|
|
23
|
+
notTested: outcomes.filter((o) => verdict(o) === 'not_tested').length,
|
|
24
|
+
oracleCorrect: judged.filter((o) => o.oracleMatch === true).length,
|
|
25
|
+
oracleWrong: judged.filter((o) => o.oracleMatch === false).length,
|
|
26
|
+
oracleUnjudged: outcomes.length - judged.length,
|
|
27
|
+
verifiedCorrect: passed.filter((o) => o.oracleMatch === true).length,
|
|
28
|
+
falseSuccess: passed.filter((o) => o.oracleMatch === false).length,
|
|
29
|
+
abstainedCorrect: withheld.filter((o) => o.oracleMatch === true).length,
|
|
30
|
+
rejectedWrong: withheld.filter((o) => o.oracleMatch === false).length,
|
|
31
|
+
structuralSuccess: outcomes.filter((o) => o.structuralSuccess).length,
|
|
32
|
+
structuralFalseSuccess: judged.filter((o) => o.structuralSuccess && o.oracleMatch === false).length,
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
/** Null rather than 0 when the denominator is empty: no cases is not a score of zero. */
|
|
36
|
+
export function ratio(numerator, denominator) {
|
|
37
|
+
return denominator === 0 ? null : numerator / denominator;
|
|
38
|
+
}
|
|
39
|
+
const pct = (value) => value === null ? 'n/a (no cases)' : `${(value * 100).toFixed(1)}%`;
|
|
40
|
+
/** Every rate prints its own denominator, so no name has to be guessed at. */
|
|
41
|
+
export function formatMatrix(m, label) {
|
|
42
|
+
return [
|
|
43
|
+
`${label}:`,
|
|
44
|
+
` passed: ${m.passed}/${m.cases}`,
|
|
45
|
+
` failed: ${m.failed}/${m.cases}`,
|
|
46
|
+
` not_tested: ${m.notTested}/${m.cases}`,
|
|
47
|
+
` false success: ${m.falseSuccess}/${m.judged} oracle-judged cases (${pct(ratio(m.falseSuccess, m.judged))})`,
|
|
48
|
+
` precision: ${m.verifiedCorrect}/${m.passed} passed cases were correct (${pct(ratio(m.verifiedCorrect, m.passed))})`,
|
|
49
|
+
` correct coverage: ${m.verifiedCorrect}/${m.oracleCorrect} correct answers were verified (${pct(ratio(m.verifiedCorrect, m.oracleCorrect))})`,
|
|
50
|
+
].join('\n');
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* One layer per check the contract has gained, so adding a check reads as a
|
|
54
|
+
* change to two numbers that move in opposite directions: how many wrong
|
|
55
|
+
* answers stopped being blessed, and how many right ones stopped being
|
|
56
|
+
* verified.
|
|
57
|
+
*/
|
|
58
|
+
export function formatLayers(outcomes, layers) {
|
|
59
|
+
const structural = tabulate(outcomes, []);
|
|
60
|
+
const lines = [];
|
|
61
|
+
lines.push(`cases: ${structural.cases}${structural.oracleUnjudged > 0 ? ` (${structural.oracleUnjudged} the oracle could not judge)` : ''}`);
|
|
62
|
+
lines.push('');
|
|
63
|
+
lines.push('structural-only (non_empty + required_fields):');
|
|
64
|
+
lines.push(` called success: ${structural.structuralSuccess}/${structural.cases}`);
|
|
65
|
+
lines.push(` false success: ${structural.structuralFalseSuccess}/${structural.judged} oracle-judged cases (${pct(ratio(structural.structuralFalseSuccess, structural.judged))})`);
|
|
66
|
+
for (const layer of layers) {
|
|
67
|
+
lines.push('');
|
|
68
|
+
lines.push(formatMatrix(tabulate(outcomes, layer.required), layer.label));
|
|
69
|
+
}
|
|
70
|
+
const last = tabulate(outcomes, layers[layers.length - 1]?.required ?? []);
|
|
71
|
+
lines.push('');
|
|
72
|
+
lines.push(`quadrants at "${layers[layers.length - 1]?.label ?? 'structural'}" (oracle-judged cases only):`);
|
|
73
|
+
lines.push(` verified_correct: ${last.verifiedCorrect}`);
|
|
74
|
+
lines.push(` false_success: ${last.falseSuccess}`);
|
|
75
|
+
lines.push(` abstained_correct: ${last.abstainedCorrect}`);
|
|
76
|
+
lines.push(` rejected_wrong: ${last.rejectedWrong}`);
|
|
77
|
+
lines.push('');
|
|
78
|
+
lines.push('per case:');
|
|
79
|
+
const width = Math.max(...outcomes.map((o) => o.id.length));
|
|
80
|
+
for (const o of outcomes) {
|
|
81
|
+
const oracle = o.oracleMatch === null ? 'unjudged' : o.oracleMatch ? 'correct' : 'wrong';
|
|
82
|
+
const verdicts = layers.map((l) => `${l.label}=${verdictFrom(o, l.required)}`).join(' ');
|
|
83
|
+
const flag = verdictFrom(o, layers[layers.length - 1]?.required ?? []) === 'passed' && o.oracleMatch === false
|
|
84
|
+
? ' <- FALSE SUCCESS' : '';
|
|
85
|
+
lines.push(` ${o.id.padEnd(width)} oracle=${oracle.padEnd(8)} ${verdicts}${flag}`);
|
|
86
|
+
}
|
|
87
|
+
return lines.join('\n');
|
|
88
|
+
}
|
|
89
|
+
/** The structural bar, kept here so the baseline cannot drift from what it claims to be. */
|
|
90
|
+
export function structurallySuccessful(items, fields) {
|
|
91
|
+
if (items.length === 0)
|
|
92
|
+
return false;
|
|
93
|
+
return !fields.some((field) => items.some((item) => {
|
|
94
|
+
const value = item[field];
|
|
95
|
+
return value === null || value === undefined || value === '';
|
|
96
|
+
}));
|
|
97
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
const NAVIGATION_TIMEOUT_MS = 30_000;
|
|
2
|
+
const SELECTOR_TIMEOUT_MS = 10_000;
|
|
3
|
+
/** After the items appear, give late XHR a moment to land before reading. */
|
|
4
|
+
export const SETTLE_MS = 800;
|
|
5
|
+
/**
|
|
6
|
+
* Loads a page and waits for its items, without requiring the network to fall
|
|
7
|
+
* silent.
|
|
8
|
+
*
|
|
9
|
+
* `networkidle` never arrives on a site that polls or streams analytics —
|
|
10
|
+
* arbeitnow.com simply timed out — and waiting for silence is the wrong
|
|
11
|
+
* condition anyway. What matters is that the items are present. The wait is for
|
|
12
|
+
* the selector, with a short settle for anything still in flight, and a
|
|
13
|
+
* network-idle attempt only as a best effort that is allowed to fail.
|
|
14
|
+
*/
|
|
15
|
+
export async function navigateAndSettle(page, url, itemSelector) {
|
|
16
|
+
await page.goto(url, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS });
|
|
17
|
+
await page
|
|
18
|
+
.waitForSelector(itemSelector, { timeout: SELECTOR_TIMEOUT_MS })
|
|
19
|
+
.catch(() => undefined);
|
|
20
|
+
await page.waitForLoadState('networkidle', { timeout: SETTLE_MS }).catch(() => undefined);
|
|
21
|
+
await page.waitForTimeout(SETTLE_MS);
|
|
22
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { costSink } from '../measurement.js';
|
|
2
|
+
import { chromium } from 'playwright';
|
|
3
|
+
import { BROWSER_USER_AGENT, observePage } from './session.js';
|
|
4
|
+
/**
|
|
5
|
+
* One browser process reused across tasks.
|
|
6
|
+
*
|
|
7
|
+
* A cold launch costs about 800ms, which on a light page is most of what a
|
|
8
|
+
* browser run costs at all. Keeping the process alive separates "this task
|
|
9
|
+
* needed a browser" from "this task paid to start one", which is the
|
|
10
|
+
* difference between avoiding a full browser and being browser-free.
|
|
11
|
+
*
|
|
12
|
+
* Each task still gets a fresh context, so cookies and storage cannot leak
|
|
13
|
+
* from one task into the next and make a recipe look better than it is.
|
|
14
|
+
*/
|
|
15
|
+
export class BrowserPool {
|
|
16
|
+
browser = null;
|
|
17
|
+
async acquire() {
|
|
18
|
+
const launched = this.browser === null;
|
|
19
|
+
this.browser ??= await chromium.launch({ headless: true });
|
|
20
|
+
if (launched)
|
|
21
|
+
costSink()({ browserLaunches: 1 });
|
|
22
|
+
const context = await this.browser.newContext({ userAgent: BROWSER_USER_AGENT });
|
|
23
|
+
const page = await context.newPage();
|
|
24
|
+
const { cost, drain } = observePage(page);
|
|
25
|
+
return { page, cost, launched, release: async () => { await context.close(); await drain(); } };
|
|
26
|
+
}
|
|
27
|
+
async close() {
|
|
28
|
+
await this.browser?.close();
|
|
29
|
+
this.browser = null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import { costSink } from '../measurement.js';
|
|
2
|
+
import { chromium } from 'playwright';
|
|
3
|
+
export const BROWSER_USER_AGENT = 'webrecipe/0.1 (+https://github.com/Pillsoon/webrecipe)';
|
|
4
|
+
/**
|
|
5
|
+
* A cold browser with instrumentation attached. Nothing is blocked or cached —
|
|
6
|
+
* this is the baseline the whole project is measured against, so it has to pay
|
|
7
|
+
* the real cost.
|
|
8
|
+
*/
|
|
9
|
+
export async function openSession() {
|
|
10
|
+
const browser = await chromium.launch({ headless: true });
|
|
11
|
+
costSink()({ browserLaunches: 1 });
|
|
12
|
+
const context = await browser.newContext({ userAgent: BROWSER_USER_AGENT });
|
|
13
|
+
const page = await context.newPage();
|
|
14
|
+
const { cost, drain } = observePage(page);
|
|
15
|
+
return { page, cost, close: async () => { await browser.close(); await drain(); } };
|
|
16
|
+
}
|
|
17
|
+
/** Count all content types; drain settled body reads after the page closes. */
|
|
18
|
+
export function observePage(page) {
|
|
19
|
+
const cost = { pageNavigations: 0, networkRequests: 0, bytesDownloaded: 0, unreadResponseBodies: 0 };
|
|
20
|
+
const charge = costSink();
|
|
21
|
+
const pending = new Set();
|
|
22
|
+
page.on('request', () => { cost.networkRequests += 1; charge({ networkRequests: 1 }); });
|
|
23
|
+
page.on('framenavigated', (frame) => {
|
|
24
|
+
if (frame === page.mainFrame()) {
|
|
25
|
+
cost.pageNavigations += 1;
|
|
26
|
+
charge({ pageNavigations: 1 });
|
|
27
|
+
}
|
|
28
|
+
});
|
|
29
|
+
page.on('response', (response) => {
|
|
30
|
+
// These responses have no HTTP body; Playwright may reject body() for them.
|
|
31
|
+
if (response.request().method() === 'HEAD' || [204, 205, 304].includes(response.status()))
|
|
32
|
+
return;
|
|
33
|
+
const read = response.body().then((buf) => {
|
|
34
|
+
cost.bytesDownloaded += buf.byteLength;
|
|
35
|
+
charge({ bytesDownloaded: buf.byteLength });
|
|
36
|
+
}).catch(() => {
|
|
37
|
+
cost.unreadResponseBodies += 1;
|
|
38
|
+
charge({ unreadResponseBodies: 1 });
|
|
39
|
+
});
|
|
40
|
+
pending.add(read);
|
|
41
|
+
void read.finally(() => pending.delete(read));
|
|
42
|
+
});
|
|
43
|
+
return { cost, drain: async () => { await Promise.all(pending); } };
|
|
44
|
+
}
|