webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
import { parseHtmlFragment } from '../executor/extract.js';
|
|
2
|
+
const SAMPLE_COUNT = 3;
|
|
3
|
+
/** How far up the tree an address may walk before the chain gets too brittle to trust. */
|
|
4
|
+
const ADDRESS_DEPTH = 8;
|
|
5
|
+
/** Text these carry is code, not content, and it would dominate the ranking. */
|
|
6
|
+
const NOISE = 'script, style, noscript, template';
|
|
7
|
+
/**
|
|
8
|
+
* The document every part of this harness works on. The generator and the
|
|
9
|
+
* scorer have to agree on what is on the page: scored against a document that
|
|
10
|
+
* still holds its `<template>` contents, a correct selector reads as
|
|
11
|
+
* over-matching and a wrong one can read as exact.
|
|
12
|
+
*/
|
|
13
|
+
export function loadPage(html) {
|
|
14
|
+
const $ = parseHtmlFragment(html);
|
|
15
|
+
$(NOISE).remove();
|
|
16
|
+
return $;
|
|
17
|
+
}
|
|
18
|
+
function nodesOf($, selector) { return $(selector).toArray(); }
|
|
19
|
+
export const isEl = (n) => 'tagName' in n;
|
|
20
|
+
export function elementsOf($, selector) {
|
|
21
|
+
return nodesOf($, selector).filter(isEl);
|
|
22
|
+
}
|
|
23
|
+
export const normalize = (value) => value.replace(/\s+/g, ' ').trim();
|
|
24
|
+
/**
|
|
25
|
+
* Everything this element says: its rendered text, and the alternative text of
|
|
26
|
+
* any image it holds. A shelf of books says nothing in text at all — each item
|
|
27
|
+
* is a cover, and the title is the cover's `alt` — so a definition that stops
|
|
28
|
+
* at `.text()` cannot tell one item from another. The generator, the label
|
|
29
|
+
* dump and the scorer all read the page through this, for the same reason they
|
|
30
|
+
* all parse it through `loadPage`.
|
|
31
|
+
*/
|
|
32
|
+
export function perceivedText($, el) {
|
|
33
|
+
const own = el.tagName.toLowerCase() === 'img' ? $(el).attr('alt') ?? '' : '';
|
|
34
|
+
const alts = $(el).find('img[alt]').map((_, img) => $(img).attr('alt') ?? '').toArray();
|
|
35
|
+
return normalize([own, $(el).text(), ...alts].join(' '));
|
|
36
|
+
}
|
|
37
|
+
const textOf = perceivedText;
|
|
38
|
+
/** A class like Tailwind's `md:flex` is not a bare CSS identifier; escape it. */
|
|
39
|
+
export function escapeIdent(value) {
|
|
40
|
+
return value.replace(/[^a-zA-Z0-9_-]/g, (ch) => `\\${ch}`);
|
|
41
|
+
}
|
|
42
|
+
export function signatureOf($, el) {
|
|
43
|
+
const classes = ($(el).attr('class') ?? '').trim().split(/\s+/).filter(Boolean);
|
|
44
|
+
return el.tagName.toLowerCase() + classes.map((c) => `.${escapeIdent(c)}`).join('');
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* The shortest selector that addresses this one element, or null when eight
|
|
48
|
+
* ancestors were not enough to find one. Uniqueness is checked by running each
|
|
49
|
+
* form rather than assumed, because an escaped class or an unusual document can
|
|
50
|
+
* make a plausible-looking address match something else.
|
|
51
|
+
*/
|
|
52
|
+
function addressOf($, el, depth = ADDRESS_DEPTH, memo = new Map()) {
|
|
53
|
+
// Groups under one parent share that parent, and parents share ancestors: on a
|
|
54
|
+
// 10k-element page the same address was recomputed, whole-document query and
|
|
55
|
+
// all, hundreds of times.
|
|
56
|
+
const known = memo.get(el)?.get(depth);
|
|
57
|
+
if (known !== undefined)
|
|
58
|
+
return known;
|
|
59
|
+
const address = uncachedAddressOf($, el, depth, memo);
|
|
60
|
+
if (!memo.has(el))
|
|
61
|
+
memo.set(el, new Map());
|
|
62
|
+
memo.get(el).set(depth, address);
|
|
63
|
+
return address;
|
|
64
|
+
}
|
|
65
|
+
function uncachedAddressOf($, el, depth, memo) {
|
|
66
|
+
if (depth === 0)
|
|
67
|
+
return null;
|
|
68
|
+
const tag = el.tagName.toLowerCase();
|
|
69
|
+
if (tag === 'body' || tag === 'html')
|
|
70
|
+
return tag;
|
|
71
|
+
const id = $(el).attr('id');
|
|
72
|
+
if (id !== undefined && id !== '') {
|
|
73
|
+
const byId = `#${escapeIdent(id)}`;
|
|
74
|
+
if (elementsOf($, byId).length === 1)
|
|
75
|
+
return byId;
|
|
76
|
+
}
|
|
77
|
+
const own = signatureOf($, el);
|
|
78
|
+
if (elementsOf($, own).length === 1)
|
|
79
|
+
return own;
|
|
80
|
+
const parent = el.parent;
|
|
81
|
+
if (parent === null || !isEl(parent))
|
|
82
|
+
return null;
|
|
83
|
+
const parentAddress = addressOf($, parent, depth - 1, memo);
|
|
84
|
+
if (parentAddress === null)
|
|
85
|
+
return null;
|
|
86
|
+
const nth = $(el).prevAll(tag).length + 1;
|
|
87
|
+
const chained = `${parentAddress} > ${tag}:nth-of-type(${nth})`;
|
|
88
|
+
return elementsOf($, chained).length === 1 ? chained : null;
|
|
89
|
+
}
|
|
90
|
+
function groupsOf($) {
|
|
91
|
+
const groups = [];
|
|
92
|
+
for (const parent of elementsOf($, '*')) {
|
|
93
|
+
const bySignature = new Map();
|
|
94
|
+
for (const child of $(parent).children().toArray().filter(isEl)) {
|
|
95
|
+
const signature = signatureOf($, child);
|
|
96
|
+
const run = bySignature.get(signature);
|
|
97
|
+
if (run)
|
|
98
|
+
run.push(child);
|
|
99
|
+
else
|
|
100
|
+
bySignature.set(signature, [child]);
|
|
101
|
+
}
|
|
102
|
+
for (const [signature, elements] of bySignature) {
|
|
103
|
+
if (elements.length >= 2)
|
|
104
|
+
groups.push({ parent, signature, elements });
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
return groups;
|
|
108
|
+
}
|
|
109
|
+
/** Materialises a proposed selector: what it matches is what gets reported. */
|
|
110
|
+
function materialize($, selector, scoped) {
|
|
111
|
+
const matched = elementsOf($, selector);
|
|
112
|
+
if (matched.length < 2)
|
|
113
|
+
return null;
|
|
114
|
+
if (matched.every((el) => textOf($, el) === ''))
|
|
115
|
+
return null;
|
|
116
|
+
const samples = matched.slice(0, SAMPLE_COUNT).map((el) => textOf($, el));
|
|
117
|
+
return { selector, count: matched.length, samples, scoped };
|
|
118
|
+
}
|
|
119
|
+
function weightOf($, selector) {
|
|
120
|
+
return elementsOf($, selector).reduce((total, el) => total + textOf($, el).length, 0);
|
|
121
|
+
}
|
|
122
|
+
export function generateCandidates(html) {
|
|
123
|
+
const $ = loadPage(html);
|
|
124
|
+
const groups = groupsOf($);
|
|
125
|
+
const proposals = new Map();
|
|
126
|
+
// The bare signature first, so that when a scoped form turns out to select
|
|
127
|
+
// the same elements the shorter one is the survivor.
|
|
128
|
+
for (const group of groups)
|
|
129
|
+
proposals.set(group.signature, false);
|
|
130
|
+
const addresses = new Map();
|
|
131
|
+
for (const group of groups) {
|
|
132
|
+
const parentAddress = addressOf($, group.parent, ADDRESS_DEPTH, addresses);
|
|
133
|
+
if (parentAddress === null)
|
|
134
|
+
continue;
|
|
135
|
+
const scoped = `${parentAddress} > ${group.signature}`;
|
|
136
|
+
const matched = new Set(elementsOf($, scoped));
|
|
137
|
+
// The scoped form has to still reach every element of the group it came
|
|
138
|
+
// from — an escape this code got wrong would show up here as a selector
|
|
139
|
+
// that no longer finds its own elements. It does not have to stop there:
|
|
140
|
+
// a class selector matches by subset while a group is one exact class set,
|
|
141
|
+
// so a carousel that marks some slides active splits twenty items into
|
|
142
|
+
// three groups whose shared selector legitimately selects all twenty.
|
|
143
|
+
// Requiring the group back exactly discarded that selector, which was the
|
|
144
|
+
// only exact candidate openlibrary had.
|
|
145
|
+
const reaches = group.elements.every((el) => matched.has(el));
|
|
146
|
+
if (reaches && matched.size >= 2 && !proposals.has(scoped))
|
|
147
|
+
proposals.set(scoped, true);
|
|
148
|
+
}
|
|
149
|
+
const built = [];
|
|
150
|
+
const seen = new Set();
|
|
151
|
+
const index = new Map();
|
|
152
|
+
elementsOf($, '*').forEach((el, i) => index.set(el, i));
|
|
153
|
+
for (const [selector, scoped] of proposals) {
|
|
154
|
+
const candidate = materialize($, selector, scoped);
|
|
155
|
+
if (candidate === null)
|
|
156
|
+
continue;
|
|
157
|
+
const identity = elementsOf($, selector).map((el) => index.get(el)).join(',');
|
|
158
|
+
if (seen.has(identity))
|
|
159
|
+
continue;
|
|
160
|
+
seen.add(identity);
|
|
161
|
+
built.push(candidate);
|
|
162
|
+
}
|
|
163
|
+
// Once per candidate, not once per comparison: the sort asked for each weight
|
|
164
|
+
// a few thousand times, and each answer read the text of every match.
|
|
165
|
+
const weights = new Map(built.map((c) => [c.selector, weightOf($, c.selector)]));
|
|
166
|
+
return built
|
|
167
|
+
.sort((a, b) => weights.get(b.selector) - weights.get(a.selector) || b.count - a.count);
|
|
168
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a learned task must prove, decided from the plan's own capabilities.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic on purpose, and deliberately unaware of any result. A contract
|
|
5
|
+
* chosen from what happened to pass would be a test whose passing grade was
|
|
6
|
+
* written after the test, which is the one thing a verification model cannot
|
|
7
|
+
* afford. Nothing here consults the benchmark's oracles either: the oracle
|
|
8
|
+
* grades this system from outside, and a contract derived from it would make
|
|
9
|
+
* the engine its own examiner.
|
|
10
|
+
*
|
|
11
|
+
* It stays this small until a check exists to justify more. A requirement for
|
|
12
|
+
* a check nobody has implemented only produces plans that can never verify.
|
|
13
|
+
*
|
|
14
|
+
* A search asks for both query checks because neither is enough alone: the
|
|
15
|
+
* first passes a site that honours the query and answers with the wrong
|
|
16
|
+
* records, and the second passes a site whose words happen to line up.
|
|
17
|
+
*
|
|
18
|
+
* Pagination is required exactly when the plan templates `page`, which is not
|
|
19
|
+
* the same as the site having pages. Templating it is an authoring act: it
|
|
20
|
+
* makes `run --page N` a supported operation, and what a task supports is what
|
|
21
|
+
* it has to be right about. A task whose promise is one page does not template
|
|
22
|
+
* page, and is not asked to prove anything about it.
|
|
23
|
+
*/
|
|
24
|
+
export function buildContract(intent, templated) {
|
|
25
|
+
const required = ['non_empty', 'required_fields'];
|
|
26
|
+
if (intent === 'search' && templated.has('query'))
|
|
27
|
+
required.push('query_honored', 'lexical_query_consistency');
|
|
28
|
+
if (templated.has('page'))
|
|
29
|
+
required.push('pagination_honored');
|
|
30
|
+
return { required };
|
|
31
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { extractBySelector, parseHtmlFragment } from '../executor/extract.js';
|
|
2
|
+
import { isEl, normalize, signatureOf } from './candidates.js';
|
|
3
|
+
const SAMPLE_COUNT = 3;
|
|
4
|
+
const MAX_CANDIDATES = 40;
|
|
5
|
+
/** How many ancestors a descendant's spec may name before it is too brittle. */
|
|
6
|
+
const SPEC_DEPTH = 3;
|
|
7
|
+
/** Attributes worth reading. Everything else on a page is styling or state. */
|
|
8
|
+
const ATTRIBUTES = ['href', 'src', 'id'];
|
|
9
|
+
/** Elements whose content is not data: script, style, noscript, template. */
|
|
10
|
+
const NOISE = new Set(['script', 'style', 'noscript', 'template']);
|
|
11
|
+
const attributesOf = ($, el) => [...ATTRIBUTES, ...Object.keys(el.attribs ?? {}).filter((a) => a.startsWith('data-'))]
|
|
12
|
+
.filter((a) => ($(el).attr(a) ?? '') !== '');
|
|
13
|
+
/**
|
|
14
|
+
* The shortest selector naming this descendant inside its item, or null when
|
|
15
|
+
* three levels were not enough to name it alone. Ambiguity is dropped rather
|
|
16
|
+
* than guessed at: with many descendants to choose from, one that cannot be
|
|
17
|
+
* addressed uniquely is not worth offering.
|
|
18
|
+
*/
|
|
19
|
+
function relativeSpec($, item, el, matches) {
|
|
20
|
+
let spec = signatureOf($, el);
|
|
21
|
+
let node = el;
|
|
22
|
+
for (let depth = 0; depth < SPEC_DEPTH; depth++) {
|
|
23
|
+
if (matches(spec) === 1)
|
|
24
|
+
return spec;
|
|
25
|
+
const parent = node.parent;
|
|
26
|
+
if (parent === null || !isEl(parent) || parent === item)
|
|
27
|
+
return null;
|
|
28
|
+
spec = `${signatureOf($, parent)} > ${spec}`;
|
|
29
|
+
node = parent;
|
|
30
|
+
}
|
|
31
|
+
return null;
|
|
32
|
+
}
|
|
33
|
+
export function fieldCandidates(html, itemSelector, limit = MAX_CANDIDATES) {
|
|
34
|
+
const $ = parseHtmlFragment(html);
|
|
35
|
+
const items = $(itemSelector).toArray().filter(isEl);
|
|
36
|
+
if (items.length === 0)
|
|
37
|
+
return [];
|
|
38
|
+
const specs = new Set();
|
|
39
|
+
for (const item of items) {
|
|
40
|
+
specs.add('');
|
|
41
|
+
for (const attr of attributesOf($, item))
|
|
42
|
+
specs.add(`@${attr}`);
|
|
43
|
+
// Descendants of one item share most signatures, and each `find` walks the
|
|
44
|
+
// whole item subtree: on a page-sized item that was 60 seconds of asking
|
|
45
|
+
// the same question. Same call, asked once per spec.
|
|
46
|
+
const counts = new Map();
|
|
47
|
+
const matches = (spec) => {
|
|
48
|
+
let n = counts.get(spec);
|
|
49
|
+
if (n === undefined) {
|
|
50
|
+
n = $(item).find(spec).length;
|
|
51
|
+
counts.set(spec, n);
|
|
52
|
+
}
|
|
53
|
+
return n;
|
|
54
|
+
};
|
|
55
|
+
for (const el of $(item).find('*').toArray().filter(isEl)) {
|
|
56
|
+
if (NOISE.has(el.tagName.toLowerCase()))
|
|
57
|
+
continue;
|
|
58
|
+
const rel = relativeSpec($, item, el, matches);
|
|
59
|
+
if (rel === null)
|
|
60
|
+
continue;
|
|
61
|
+
specs.add(rel);
|
|
62
|
+
for (const attr of attributesOf($, el))
|
|
63
|
+
specs.add(`${rel}@${attr}`);
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
// Filter out specs with escaped @ before extraction, as they would be misparsed by the engine
|
|
67
|
+
const validSpecs = [...specs].filter((spec) => !spec.includes('\\@'));
|
|
68
|
+
// One extraction with every spec as its own field: the values are then read
|
|
69
|
+
// exactly as the engine would read them, in a single parse.
|
|
70
|
+
const named = validSpecs.map((spec, i) => [`f${i}`, spec]);
|
|
71
|
+
const rows = extractBySelector(html, itemSelector, Object.fromEntries(named));
|
|
72
|
+
const built = named.map(([name, spec]) => {
|
|
73
|
+
const values = rows.map((row) => normalize(String(row[name] ?? '')));
|
|
74
|
+
const present = values.filter((v) => v !== '');
|
|
75
|
+
return {
|
|
76
|
+
spec,
|
|
77
|
+
coverage: present.length / values.length,
|
|
78
|
+
distinct: new Set(present).size / values.length,
|
|
79
|
+
samples: present.slice(0, SAMPLE_COUNT),
|
|
80
|
+
};
|
|
81
|
+
});
|
|
82
|
+
return built
|
|
83
|
+
.filter((c) => c.coverage > 0)
|
|
84
|
+
.sort((a, b) => b.coverage - a.coverage || b.distinct - a.distinct || a.spec.length - b.spec.length)
|
|
85
|
+
.slice(0, limit);
|
|
86
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { captureDom } from './snapshot.js';
|
|
2
|
+
import { generateCandidates } from './candidates.js';
|
|
3
|
+
import { fieldCandidates } from './fields.js';
|
|
4
|
+
/**
|
|
5
|
+
* One browser visit, turned into the options an agent chooses between.
|
|
6
|
+
*
|
|
7
|
+
* Nothing is stored and nothing is decided. The ranking is the order of a list
|
|
8
|
+
* somebody else reads: the item-selector heuristic put the right answer first
|
|
9
|
+
* on none of six labelled pages and second or third on five of them, which is
|
|
10
|
+
* a bad ranker and a fine shortlist. The one site tried outside that corpus,
|
|
11
|
+
* Hacker News, put the answer seventh, so the default has to reach past
|
|
12
|
+
* "second or third" to still be useful off the labelled set.
|
|
13
|
+
*/
|
|
14
|
+
const DEFAULT_DEPTH = 10;
|
|
15
|
+
const FIELDS_PER_ITEM = 12;
|
|
16
|
+
export function learnFromHtml(url, html, depth = DEFAULT_DEPTH) {
|
|
17
|
+
const items = generateCandidates(html)
|
|
18
|
+
.slice(0, depth)
|
|
19
|
+
.map((candidate) => ({
|
|
20
|
+
candidate,
|
|
21
|
+
fields: fieldCandidates(html, candidate.selector, FIELDS_PER_ITEM),
|
|
22
|
+
}));
|
|
23
|
+
return { url, items };
|
|
24
|
+
}
|
|
25
|
+
export async function learn(url, depth = DEFAULT_DEPTH) {
|
|
26
|
+
return learnFromHtml(url, (await captureDom(url)).html, depth);
|
|
27
|
+
}
|
|
28
|
+
const pct = (v) => v.toFixed(2);
|
|
29
|
+
/**
|
|
30
|
+
* Every line here is a fragment of a command someone will run, so a spec has
|
|
31
|
+
* to survive being pasted into one. A class carrying `$` or a backtick, which
|
|
32
|
+
* this generator escapes and therefore can produce, is mangled by a
|
|
33
|
+
* double-quoted shell; single quotes are the encoding that matches the context.
|
|
34
|
+
*/
|
|
35
|
+
export const shell = (value) => `'${value.replaceAll("'", String.raw `'\''`)}'`;
|
|
36
|
+
/** Visibly cut, so a value that diverges past the cut cannot look identical. */
|
|
37
|
+
const clip = (value, width) => value.length <= width ? value : `${value.slice(0, width - 1)}…`;
|
|
38
|
+
export function formatLearn(report) {
|
|
39
|
+
const lines = [report.url, ''];
|
|
40
|
+
report.items.forEach(({ candidate, fields }, i) => {
|
|
41
|
+
lines.push(`${i + 1}. --items ${shell(candidate.selector)} ${candidate.count} items`);
|
|
42
|
+
if (candidate.samples[0] !== undefined)
|
|
43
|
+
lines.push(` sample: ${candidate.samples[0].slice(0, 90)}`);
|
|
44
|
+
for (const f of fields) {
|
|
45
|
+
lines.push(` --field NAME=${shell(f.spec).padEnd(34)} ` +
|
|
46
|
+
`cover ${pct(f.coverage)} distinct ${pct(f.distinct)} ${clip(f.samples[0] ?? '', 60)}`);
|
|
47
|
+
}
|
|
48
|
+
lines.push('');
|
|
49
|
+
});
|
|
50
|
+
return lines.join('\n');
|
|
51
|
+
}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { atomicWrite, siteName, intentName, publicUrl, PlanVerificationSchema } from '../local.js';
|
|
2
|
+
import { mkdir, readFile, readdir } from 'node:fs/promises';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { z } from 'zod';
|
|
5
|
+
import { renderUrlTemplate } from '../recipes/template.js';
|
|
6
|
+
/**
|
|
7
|
+
* A plan an agent taught, as a file.
|
|
8
|
+
*
|
|
9
|
+
* `benchmark/plans.ts` is the benchmark's fixed input and must not start
|
|
10
|
+
* varying with what anyone has learned, so learned plans live apart and carry
|
|
11
|
+
* their own origin. That is what removes the need to edit two TypeScript files
|
|
12
|
+
* before the engine can visit a site it has never seen.
|
|
13
|
+
*/
|
|
14
|
+
export const LearnedPlanSchema = z.object({
|
|
15
|
+
origin: z.string().url(),
|
|
16
|
+
/** Rendered against the task's input, so one plan serves every input. */
|
|
17
|
+
urlTemplate: z.string().min(1),
|
|
18
|
+
itemSelector: z.string().min(1),
|
|
19
|
+
fields: z.record(z.string()).refine((f) => Object.keys(f).length > 0, 'a plan with no fields extracts nothing'),
|
|
20
|
+
/**
|
|
21
|
+
* What this task must prove, and what it has proven. Optional because plans
|
|
22
|
+
* taught before contracts existed must keep loading; without one a run is
|
|
23
|
+
* reported as structural and never as verified.
|
|
24
|
+
*/
|
|
25
|
+
verification: PlanVerificationSchema.optional(),
|
|
26
|
+
});
|
|
27
|
+
const INTENTS = ['search', 'list', 'detail'];
|
|
28
|
+
export async function saveLearnedPlan(dir, site, intent, plan) {
|
|
29
|
+
siteName(site);
|
|
30
|
+
intentName(intent);
|
|
31
|
+
const parsed = LearnedPlanSchema.parse(plan);
|
|
32
|
+
publicUrl(parsed.origin);
|
|
33
|
+
await mkdir(join(dir, site), { recursive: true });
|
|
34
|
+
const path = join(dir, site, `${intent}.json`);
|
|
35
|
+
await atomicWrite(path, `${JSON.stringify(parsed, null, 2)}\n`);
|
|
36
|
+
return path;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Hand-written plans with learned ones layered over them, per intent.
|
|
40
|
+
*
|
|
41
|
+
* Merging at the site key instead would drop every hand-written intent for a
|
|
42
|
+
* site the moment one of its intents was taught.
|
|
43
|
+
*/
|
|
44
|
+
export function mergePlans(hand, learned) {
|
|
45
|
+
const merged = { ...hand };
|
|
46
|
+
for (const [site, intents] of Object.entries(learned)) {
|
|
47
|
+
merged[site] = { ...merged[site], ...intents };
|
|
48
|
+
}
|
|
49
|
+
return merged;
|
|
50
|
+
}
|
|
51
|
+
export async function loadLearnedPlans(dir) {
|
|
52
|
+
const sites = await readdir(dir, { withFileTypes: true }).catch((error) => { if (error.code === 'ENOENT')
|
|
53
|
+
return []; throw error; });
|
|
54
|
+
const plans = {};
|
|
55
|
+
const origins = {};
|
|
56
|
+
const verifications = {};
|
|
57
|
+
const rawPlans = {};
|
|
58
|
+
for (const entry of sites) {
|
|
59
|
+
if (!entry.isDirectory())
|
|
60
|
+
continue;
|
|
61
|
+
for (const intent of INTENTS) {
|
|
62
|
+
const path = join(dir, entry.name, `${intent}.json`);
|
|
63
|
+
const raw = await readFile(path, 'utf8').catch((error) => { if (error.code === 'ENOENT')
|
|
64
|
+
return null; throw error; });
|
|
65
|
+
if (raw === null)
|
|
66
|
+
continue;
|
|
67
|
+
let plan;
|
|
68
|
+
try {
|
|
69
|
+
plan = LearnedPlanSchema.parse(JSON.parse(raw));
|
|
70
|
+
}
|
|
71
|
+
catch (cause) {
|
|
72
|
+
throw new Error(`corrupt learned plan at ${path}`, { cause });
|
|
73
|
+
}
|
|
74
|
+
publicUrl(plan.origin);
|
|
75
|
+
if (origins[entry.name] && origins[entry.name] !== plan.origin)
|
|
76
|
+
throw new Error(`conflicting origins for ${entry.name}`);
|
|
77
|
+
origins[entry.name] = plan.origin;
|
|
78
|
+
if (plan.verification)
|
|
79
|
+
verifications[entry.name] = { ...verifications[entry.name], [intent]: plan.verification };
|
|
80
|
+
rawPlans[entry.name] = { ...rawPlans[entry.name], [intent]: plan };
|
|
81
|
+
plans[entry.name] = {
|
|
82
|
+
...plans[entry.name],
|
|
83
|
+
[intent]: {
|
|
84
|
+
url: (origin, task) => renderUrlTemplate(origin, plan.urlTemplate, task.input),
|
|
85
|
+
sameOriginOnly: true,
|
|
86
|
+
itemSelector: plan.itemSelector,
|
|
87
|
+
fields: plan.fields,
|
|
88
|
+
},
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
return { plans, origins, verifications, raw: rawPlans };
|
|
93
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { SETTLE_MS } from '../browser/navigate.js';
|
|
2
|
+
import { openSession } from '../browser/session.js';
|
|
3
|
+
const NAVIGATION_TIMEOUT_MS = 30_000;
|
|
4
|
+
/**
|
|
5
|
+
* One rendered DOM, saved so that a hand-written label keeps its meaning. A
|
|
6
|
+
* label attached to a live page rots when the site edits its markup, and two of
|
|
7
|
+
* the sites worth labelling already refuse this machine.
|
|
8
|
+
*/
|
|
9
|
+
export async function captureDom(url) {
|
|
10
|
+
const session = await openSession();
|
|
11
|
+
try {
|
|
12
|
+
await session.page.goto(url, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS });
|
|
13
|
+
await session.page
|
|
14
|
+
.waitForLoadState('networkidle', { timeout: NAVIGATION_TIMEOUT_MS })
|
|
15
|
+
.catch(() => undefined);
|
|
16
|
+
await session.page.waitForTimeout(SETTLE_MS);
|
|
17
|
+
return { html: await session.page.content(), capturedAt: new Date().toISOString() };
|
|
18
|
+
}
|
|
19
|
+
finally {
|
|
20
|
+
await session.close();
|
|
21
|
+
}
|
|
22
|
+
}
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import { siteName, intentName, publicUrl, requireReadable } from '../local.js';
|
|
2
|
+
import { browserItemsOf } from '../compiler/verify.js';
|
|
3
|
+
import { record } from '../recorder/index.js';
|
|
4
|
+
import { HeuristicCompiler, templatePath, templatedInputs } from '../compiler/heuristic.js';
|
|
5
|
+
import { compileHtmlRecipe } from '../compiler/html.js';
|
|
6
|
+
import { isRefused } from '../compiler/types.js';
|
|
7
|
+
import { RecipeRegistry } from '../recipes/registry.js';
|
|
8
|
+
import { StaticSiteResolver } from '../sites.js';
|
|
9
|
+
import { LearnedPlanSchema, saveLearnedPlan } from './plans.js';
|
|
10
|
+
import { buildContract } from './contract.js';
|
|
11
|
+
import { runQueryProbes, verifyQueryHonored } from '../verification/query-honored.js';
|
|
12
|
+
import { verifyLexicalConsistency } from '../verification/lexical-consistency.js';
|
|
13
|
+
import { runPaginationProbes, verifyPaginationHonored } from '../verification/pagination-honored.js';
|
|
14
|
+
import { Probes, runBaseline } from '../verification/probes.js';
|
|
15
|
+
import { renderUrlTemplate } from '../recipes/template.js';
|
|
16
|
+
import { PolitenessLayer } from '../net/politeness.js';
|
|
17
|
+
export async function teach(opts) {
|
|
18
|
+
siteName(opts.site);
|
|
19
|
+
intentName(opts.intent);
|
|
20
|
+
const url = publicUrl(opts.url);
|
|
21
|
+
const path = templatePath(url.pathname, opts.input);
|
|
22
|
+
const query = [...url.searchParams.entries()]
|
|
23
|
+
.map(([k, v]) => `${encodeURIComponent(k)}=${encodeURIComponent(templatePath(v, opts.input)).replace(/%7B%7B(\w+)%7D%7D/g, '{{$1}}')}`)
|
|
24
|
+
.join('&');
|
|
25
|
+
const urlTemplate = `${path}${query === '' ? '' : `?${query}`}`;
|
|
26
|
+
// A url template that does not carry an input as a placeholder returns the
|
|
27
|
+
// recorded page for every value of that input, forever, with validation
|
|
28
|
+
// passing every time — the plan itself would be the broken thing, unlike a
|
|
29
|
+
// refusal from the compiler that just means the browser fallback must answer.
|
|
30
|
+
const templated = templatedInputs([urlTemplate]);
|
|
31
|
+
const missing = Object.entries(opts.input)
|
|
32
|
+
.filter(([, v]) => String(v) !== '')
|
|
33
|
+
.map(([name]) => name)
|
|
34
|
+
.filter((name) => !templated.has(name));
|
|
35
|
+
if (missing.length > 0) {
|
|
36
|
+
throw new Error(`urlTemplate "${urlTemplate}" does not template input(s): ${missing.join(', ')}`);
|
|
37
|
+
}
|
|
38
|
+
// The contract is fixed from what the plan can vary, before a single result
|
|
39
|
+
// is seen. Evidence is filled in below by whatever actually ran.
|
|
40
|
+
const contract = buildContract(opts.intent, templated);
|
|
41
|
+
const plan = LearnedPlanSchema.parse({
|
|
42
|
+
origin: url.origin,
|
|
43
|
+
urlTemplate,
|
|
44
|
+
itemSelector: opts.itemSelector,
|
|
45
|
+
fields: opts.fields,
|
|
46
|
+
verification: { contract, evidence: {} },
|
|
47
|
+
});
|
|
48
|
+
const browserPlan = {
|
|
49
|
+
url: () => opts.url,
|
|
50
|
+
itemSelector: plan.itemSelector,
|
|
51
|
+
fields: plan.fields,
|
|
52
|
+
};
|
|
53
|
+
const sites = new StaticSiteResolver({ [opts.site]: plan.origin });
|
|
54
|
+
const task = { id: 'teach', site: opts.site, intent: opts.intent, input: opts.input };
|
|
55
|
+
const trace = await record(browserPlan, task, sites);
|
|
56
|
+
try {
|
|
57
|
+
const items = browserItemsOf(trace, browserPlan);
|
|
58
|
+
requireReadable(items, Object.keys(plan.fields));
|
|
59
|
+
// Only checks something actually ran are recorded. A required check with no
|
|
60
|
+
// entry reads back as not_tested, which is the truth; writing that status
|
|
61
|
+
// here would store an outcome no verifier produced.
|
|
62
|
+
plan.verification = {
|
|
63
|
+
contract,
|
|
64
|
+
evidence: { non_empty: { status: 'passed' }, required_fields: { status: 'passed' } },
|
|
65
|
+
};
|
|
66
|
+
const refusals = [];
|
|
67
|
+
let recipe = null;
|
|
68
|
+
const json = await new HeuristicCompiler(browserPlan).compile(trace);
|
|
69
|
+
if (isRefused(json))
|
|
70
|
+
refusals.push(`json: ${json.refused}`);
|
|
71
|
+
else
|
|
72
|
+
recipe = json;
|
|
73
|
+
if (recipe === null) {
|
|
74
|
+
const html = compileHtmlRecipe(trace, browserPlan);
|
|
75
|
+
if (isRefused(html))
|
|
76
|
+
refusals.push(`html: ${html.refused}`);
|
|
77
|
+
else
|
|
78
|
+
recipe = html;
|
|
79
|
+
}
|
|
80
|
+
// A recipe reaching past the site's own host is refused rather than stored.
|
|
81
|
+
// It is tolerable while every recipe is compiled locally from a site the
|
|
82
|
+
// caller chose, and it is not tolerable once recipes are shared, so the rule
|
|
83
|
+
// is set before anything depends on the looser one.
|
|
84
|
+
if (recipe !== null && recipe.request.origin !== undefined && recipe.request.origin !== plan.origin) {
|
|
85
|
+
refusals.push(`recipe reaches ${recipe.request.origin}, outside ${plan.origin}`);
|
|
86
|
+
recipe = null;
|
|
87
|
+
}
|
|
88
|
+
// Probing needs a recipe to judge and a contract that asks for something.
|
|
89
|
+
// It reads the compiled recipe and writes nothing, so a failed probe leaves
|
|
90
|
+
// the plan and the registry exactly as a skipped one does.
|
|
91
|
+
const wantsQuery = contract.required.includes('query_honored');
|
|
92
|
+
const wantsPagination = contract.required.includes('pagination_honored');
|
|
93
|
+
if (recipe !== null && (wantsQuery || wantsPagination) && !opts.skipSemanticVerification) {
|
|
94
|
+
const probes = new Probes({
|
|
95
|
+
recipe,
|
|
96
|
+
browserPlan: {
|
|
97
|
+
url: (origin, probe) => renderUrlTemplate(origin, plan.urlTemplate, probe.input),
|
|
98
|
+
sameOriginOnly: true,
|
|
99
|
+
itemSelector: plan.itemSelector,
|
|
100
|
+
fields: plan.fields,
|
|
101
|
+
},
|
|
102
|
+
site: opts.site,
|
|
103
|
+
intent: opts.intent,
|
|
104
|
+
input: opts.input,
|
|
105
|
+
net: new PolitenessLayer({ minIntervalMs: opts.minIntervalMs }),
|
|
106
|
+
sites,
|
|
107
|
+
});
|
|
108
|
+
// One baseline for every check. Neither check may depend on the other
|
|
109
|
+
// having run: a list task that pages but takes no query still has to be
|
|
110
|
+
// able to prove its pagination.
|
|
111
|
+
const baseline = await runBaseline(probes, opts.input, wantsQuery ? 'query' : 'page', items);
|
|
112
|
+
if (wantsQuery) {
|
|
113
|
+
const session = await runQueryProbes(probes, baseline, 'query');
|
|
114
|
+
plan.verification.evidence.query_honored = verifyQueryHonored(session);
|
|
115
|
+
plan.verification.evidence.lexical_query_consistency = verifyLexicalConsistency(session);
|
|
116
|
+
}
|
|
117
|
+
if (wantsPagination) {
|
|
118
|
+
plan.verification.evidence.pagination_honored = verifyPaginationHonored(await runPaginationProbes(probes, baseline, 'page'));
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
// A refused compile must not leave an earlier task's recipe active.
|
|
122
|
+
const registry = new RecipeRegistry(opts.recipeDir);
|
|
123
|
+
await registry.remove(opts.site, opts.intent);
|
|
124
|
+
const planPath = await saveLearnedPlan(opts.planDir, opts.site, opts.intent, plan);
|
|
125
|
+
await registry.clearLearningFailure(opts.site, opts.intent);
|
|
126
|
+
const refused = recipe === null ? refusals.join('; ') : null;
|
|
127
|
+
if (recipe !== null)
|
|
128
|
+
await registry.save(recipe);
|
|
129
|
+
else
|
|
130
|
+
await registry.setLearningFailure(opts.site, opts.intent, refused);
|
|
131
|
+
return { plan, planPath, recipe, refused, sample: items.slice(0, 3) };
|
|
132
|
+
}
|
|
133
|
+
finally {
|
|
134
|
+
await trace.dispose?.();
|
|
135
|
+
}
|
|
136
|
+
}
|