webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Per-site oracle defaults. A site absent here is graded `golden`, which is
|
|
3
|
+
* what every run before this contract used.
|
|
4
|
+
*
|
|
5
|
+
* `volatility` is documentation: it records why a mode was chosen, so a later
|
|
6
|
+
* reader can tell a considered choice from a convenient one.
|
|
7
|
+
*/
|
|
8
|
+
export const ORACLES = {
|
|
9
|
+
// Job boards: the whole listing turns over within minutes, so a stored
|
|
10
|
+
// capture measures its own age rather than the engine.
|
|
11
|
+
'remoteok.com': {
|
|
12
|
+
mode: 'paired-live',
|
|
13
|
+
volatility: 'high',
|
|
14
|
+
compare: { ordering: 'ignore', fields: ['id', 'url'] },
|
|
15
|
+
},
|
|
16
|
+
'arbeitnow.com': {
|
|
17
|
+
mode: 'paired-live',
|
|
18
|
+
volatility: 'high',
|
|
19
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
20
|
+
// The one site measured so far that can silently ignore its own search
|
|
21
|
+
// input: the parameter is discarded on redirect and every query returns the
|
|
22
|
+
// same unfiltered front page. Declared here and nowhere else, because a
|
|
23
|
+
// rule that fires on sites it was not designed for is an oracle bug.
|
|
24
|
+
semantics: { input: 'query', fields: ['title'], match: 'contains-token', minShare: 0.5 },
|
|
25
|
+
},
|
|
26
|
+
// A marketplace browse page and a small forum: the set moves over days, and
|
|
27
|
+
// both reorder equally-ranked items between requests.
|
|
28
|
+
'itch.io': {
|
|
29
|
+
mode: 'paired-live',
|
|
30
|
+
volatility: 'medium',
|
|
31
|
+
compare: { ordering: 'ignore', fields: ['id', 'title', 'url'] },
|
|
32
|
+
},
|
|
33
|
+
'tildes.net': {
|
|
34
|
+
mode: 'paired-live',
|
|
35
|
+
volatility: 'medium',
|
|
36
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
37
|
+
},
|
|
38
|
+
// Registries and documentation: a version bump is the only churn, and result
|
|
39
|
+
// ranking is stable enough to hold to.
|
|
40
|
+
'docs.rs': { mode: 'golden', volatility: 'low' },
|
|
41
|
+
'crates.io': { mode: 'golden', volatility: 'low' },
|
|
42
|
+
'hex.pm': { mode: 'golden', volatility: 'low' },
|
|
43
|
+
'jsr.io': { mode: 'golden', volatility: 'low' },
|
|
44
|
+
'pkg.go.dev': {
|
|
45
|
+
mode: 'golden',
|
|
46
|
+
volatility: 'low',
|
|
47
|
+
// Ranks ties differently between requests; the set is stable, the order is not.
|
|
48
|
+
compare: { ordering: 'ignore' },
|
|
49
|
+
},
|
|
50
|
+
'hn.algolia.com': { mode: 'golden', volatility: 'low' },
|
|
51
|
+
'flathub.org': {
|
|
52
|
+
mode: 'golden',
|
|
53
|
+
volatility: 'medium',
|
|
54
|
+
compare: { ordering: 'ignore' },
|
|
55
|
+
},
|
|
56
|
+
// Held-out 3. Four of the five browse axes turn over on their own schedule —
|
|
57
|
+
// a forum's tag feed, a federated community, a music catalog's discover
|
|
58
|
+
// listing, and a trending chart are all volatile by construction — so they
|
|
59
|
+
// are graded against the adjacent browser run rather than a stored capture.
|
|
60
|
+
'meta.discourse.org': {
|
|
61
|
+
mode: 'paired-live', volatility: 'high',
|
|
62
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
63
|
+
},
|
|
64
|
+
'lemmy.world': {
|
|
65
|
+
mode: 'paired-live', volatility: 'high',
|
|
66
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
67
|
+
},
|
|
68
|
+
'bandcamp.com': {
|
|
69
|
+
mode: 'paired-live', volatility: 'high',
|
|
70
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
71
|
+
},
|
|
72
|
+
'openlibrary.org': {
|
|
73
|
+
mode: 'paired-live', volatility: 'high',
|
|
74
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
75
|
+
},
|
|
76
|
+
// The one held-out 3 axis that does not move: a format listing is a catalog,
|
|
77
|
+
// not a feed.
|
|
78
|
+
'loc.gov': {
|
|
79
|
+
mode: 'golden', volatility: 'low',
|
|
80
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
81
|
+
},
|
|
82
|
+
// Held-out 4. The three feeds turn over continuously — two fediverse tag
|
|
83
|
+
// timelines and a forum's tag feed — so they are graded against the adjacent
|
|
84
|
+
// browser run. pixelfed's grid carries no text of any kind, so `url` is the
|
|
85
|
+
// only field there is to compare.
|
|
86
|
+
'mastodon.social': {
|
|
87
|
+
mode: 'paired-live', volatility: 'high',
|
|
88
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
89
|
+
},
|
|
90
|
+
'pixelfed.social': {
|
|
91
|
+
mode: 'paired-live', volatility: 'high',
|
|
92
|
+
compare: { ordering: 'ignore', fields: ['url'] },
|
|
93
|
+
},
|
|
94
|
+
'dev.to': {
|
|
95
|
+
mode: 'paired-live', volatility: 'high',
|
|
96
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
97
|
+
},
|
|
98
|
+
// A catalog and a registry: a discography and a search ranking both hold
|
|
99
|
+
// still long enough for a stored capture to be the scorer. `artist` is graded
|
|
100
|
+
// on musicbrainz because the composed credit column is what slot C measures.
|
|
101
|
+
'musicbrainz.org': {
|
|
102
|
+
mode: 'golden', volatility: 'low',
|
|
103
|
+
compare: { ordering: 'ignore', fields: ['title', 'url', 'artist'] },
|
|
104
|
+
},
|
|
105
|
+
'npmjs.com': {
|
|
106
|
+
mode: 'golden', volatility: 'low',
|
|
107
|
+
compare: { ordering: 'ignore', fields: ['title', 'url'] },
|
|
108
|
+
},
|
|
109
|
+
// Fixtures serve a fixed dataset in a fixed order, which makes them the only
|
|
110
|
+
// sites where ordering is answerable at all — every wild site measured so far
|
|
111
|
+
// reorders equally-ranked results between requests.
|
|
112
|
+
siteA: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
|
|
113
|
+
siteB: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
|
|
114
|
+
siteC: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
|
|
115
|
+
siteCoalesce: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
|
|
116
|
+
// Paired-live so that an unusable baseline reaches baselineValidity; under
|
|
117
|
+
// golden the stored capture is the scorer and never consults the browser.
|
|
118
|
+
siteBroken: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
|
|
119
|
+
// The one place orderingAgreement is reachable: paired-live plus a stable order.
|
|
120
|
+
siteOrdered: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'strict' } },
|
|
121
|
+
// Paired-live because the whole point is that the browser side stays healthy
|
|
122
|
+
// while the engine's client is refused; a stored golden never consults the
|
|
123
|
+
// browser, so it could not show the difference.
|
|
124
|
+
siteRefusing: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
|
|
125
|
+
// Paired-live for siteRefusing's reason, and because the browser side is the
|
|
126
|
+
// half that stays healthy here: it is the engine's client alone that is sent
|
|
127
|
+
// a different page.
|
|
128
|
+
siteStub: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
|
|
129
|
+
};
|
|
@@ -0,0 +1,436 @@
|
|
|
1
|
+
const enc = (v) => encodeURIComponent(String(v));
|
|
2
|
+
/** Hoisted because siteRefusing serves this exact page; the two must not drift. */
|
|
3
|
+
const SITE_A_PLANS = {
|
|
4
|
+
search: {
|
|
5
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
6
|
+
itemSelector: 'li.result',
|
|
7
|
+
fields: { id: '@data-id', title: 'a.title', url: 'a.title@href' },
|
|
8
|
+
},
|
|
9
|
+
list: {
|
|
10
|
+
url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
|
|
11
|
+
itemSelector: 'li.result',
|
|
12
|
+
fields: { id: '@data-id', title: 'a.title', url: 'a.title@href' },
|
|
13
|
+
},
|
|
14
|
+
detail: {
|
|
15
|
+
url: (o, t) => `${o}/item/${enc(t.input.id)}`,
|
|
16
|
+
itemSelector: 'article#detail',
|
|
17
|
+
fields: { id: '@data-id', title: 'h1.title', author: 'span.author' },
|
|
18
|
+
},
|
|
19
|
+
};
|
|
20
|
+
/**
|
|
21
|
+
* Per-site knowledge of where to go and what to read — exactly what a learned
|
|
22
|
+
* recipe replaces. Fixture plans are exact; wild plans are verified against the
|
|
23
|
+
* live pages before any benchmark number is trusted.
|
|
24
|
+
*/
|
|
25
|
+
export const PLANS = {
|
|
26
|
+
siteA: SITE_A_PLANS,
|
|
27
|
+
// siteB's API returns no url; the page composes it from the id, so a recipe
|
|
28
|
+
// can only match the browser by learning that derivation.
|
|
29
|
+
siteB: {
|
|
30
|
+
search: {
|
|
31
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
32
|
+
itemSelector: 'li.result',
|
|
33
|
+
fields: { id: '@data-id', title: '', url: 'a.permalink@href' },
|
|
34
|
+
},
|
|
35
|
+
list: {
|
|
36
|
+
url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
|
|
37
|
+
itemSelector: 'li.result',
|
|
38
|
+
fields: { id: '@data-id', title: '', url: 'a.permalink@href' },
|
|
39
|
+
},
|
|
40
|
+
detail: {
|
|
41
|
+
url: (o, t) => `${o}/item/${enc(t.input.id)}`,
|
|
42
|
+
itemSelector: 'article#detail',
|
|
43
|
+
fields: { id: '@data-id', title: 'h1.title', author: 'span.author' },
|
|
44
|
+
},
|
|
45
|
+
},
|
|
46
|
+
siteC: {
|
|
47
|
+
search: {
|
|
48
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
49
|
+
itemSelector: 'li.result',
|
|
50
|
+
fields: { id: '@data-id', title: '' },
|
|
51
|
+
},
|
|
52
|
+
list: {
|
|
53
|
+
url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
|
|
54
|
+
itemSelector: 'li.result',
|
|
55
|
+
fields: { id: '@data-id', title: '' },
|
|
56
|
+
},
|
|
57
|
+
detail: {
|
|
58
|
+
url: (o, t) => `${o}/item/${enc(t.input.id)}`,
|
|
59
|
+
itemSelector: 'article#detail',
|
|
60
|
+
fields: { id: '@data-id', title: '' },
|
|
61
|
+
},
|
|
62
|
+
},
|
|
63
|
+
// crates.io is a Svelte app; its generated class names (svelte-10p0ryf) change
|
|
64
|
+
// on every build, so only the semantic ones are usable as selectors.
|
|
65
|
+
'crates.io': {
|
|
66
|
+
search: {
|
|
67
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
68
|
+
itemSelector: '.crate-row',
|
|
69
|
+
fields: { title: 'a.name', url: 'a.name@href' },
|
|
70
|
+
},
|
|
71
|
+
detail: {
|
|
72
|
+
url: (o, t) => `${o}/crates/${enc(t.input.id)}`,
|
|
73
|
+
itemSelector: 'h1.heading',
|
|
74
|
+
fields: { title: '' },
|
|
75
|
+
},
|
|
76
|
+
},
|
|
77
|
+
'pkg.go.dev': {
|
|
78
|
+
search: {
|
|
79
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
80
|
+
itemSelector: '.SearchSnippet',
|
|
81
|
+
fields: { title: 'h2 a', url: 'h2 a@href' },
|
|
82
|
+
},
|
|
83
|
+
detail: {
|
|
84
|
+
url: (o, t) => `${o}/${String(t.input.id)}`,
|
|
85
|
+
itemSelector: '.UnitHeader-titleHeading',
|
|
86
|
+
fields: { title: '' },
|
|
87
|
+
},
|
|
88
|
+
},
|
|
89
|
+
'hn.algolia.com': {
|
|
90
|
+
search: {
|
|
91
|
+
url: (o, t) => `${o}/?query=${enc(t.input.query)}`,
|
|
92
|
+
itemSelector: '.Story',
|
|
93
|
+
fields: { title: '.Story_title a', url: '.Story_title a@href' },
|
|
94
|
+
},
|
|
95
|
+
},
|
|
96
|
+
// ---------------------------------------------------------------------------
|
|
97
|
+
// Held-out sites. Selectors were authored from the rendered DOM, as any user
|
|
98
|
+
// of this tool would; their API payloads were deliberately not inspected.
|
|
99
|
+
// ---------------------------------------------------------------------------
|
|
100
|
+
'hex.pm': {
|
|
101
|
+
search: {
|
|
102
|
+
url: (o, t) => `${o}/packages?search=${enc(t.input.query)}`,
|
|
103
|
+
itemSelector: 'li:has(a[href^="/packages/"])',
|
|
104
|
+
fields: { title: 'a[href^="/packages/"]', url: 'a[href^="/packages/"]@href' },
|
|
105
|
+
},
|
|
106
|
+
detail: {
|
|
107
|
+
url: (o, t) => `${o}/packages/${enc(t.input.id)}`,
|
|
108
|
+
itemSelector: 'h1',
|
|
109
|
+
fields: { title: '' },
|
|
110
|
+
},
|
|
111
|
+
},
|
|
112
|
+
'docs.rs': {
|
|
113
|
+
search: {
|
|
114
|
+
url: (o, t) => `${o}/releases/search?query=${enc(t.input.query)}`,
|
|
115
|
+
itemSelector: 'a.release',
|
|
116
|
+
fields: { title: '.name', description: '.description', url: '@href' },
|
|
117
|
+
},
|
|
118
|
+
},
|
|
119
|
+
'flathub.org': {
|
|
120
|
+
search: {
|
|
121
|
+
url: (o, t) => `${o}/apps/search?q=${enc(t.input.query)}`,
|
|
122
|
+
// Scoped to main: the same anchor shape appears in the nav and the
|
|
123
|
+
// recommendation strip, which would pad the result set with five rows
|
|
124
|
+
// the search API never returned.
|
|
125
|
+
itemSelector: 'main a[href^="/en/apps/"]',
|
|
126
|
+
fields: { title: 'span.truncate', url: '@href' },
|
|
127
|
+
},
|
|
128
|
+
detail: {
|
|
129
|
+
url: (o, t) => `${o}/apps/${String(t.input.id)}`,
|
|
130
|
+
itemSelector: 'h1',
|
|
131
|
+
fields: { title: '' },
|
|
132
|
+
},
|
|
133
|
+
},
|
|
134
|
+
'jsr.io': {
|
|
135
|
+
search: {
|
|
136
|
+
url: (o, t) => `${o}/packages?search=${enc(t.input.query)}`,
|
|
137
|
+
itemSelector: 'li:has(a[href^="/@"])',
|
|
138
|
+
fields: { title: 'a[href^="/@"]', url: 'a[href^="/@"]@href' },
|
|
139
|
+
},
|
|
140
|
+
},
|
|
141
|
+
// ---------------------------------------------------------------------------
|
|
142
|
+
// Held-out 2. Same rule as before: selectors authored from the rendered DOM,
|
|
143
|
+
// payloads never inspected.
|
|
144
|
+
// ---------------------------------------------------------------------------
|
|
145
|
+
// No list intent for these three: none of them paginates through a URL
|
|
146
|
+
// parameter. tildes uses an after= cursor, itch.io and arbeitnow ignore
|
|
147
|
+
// ?page= entirely, and remoteok scrolls. A recipe replays one static request,
|
|
148
|
+
// so a second page is not expressible — a design limit, not a missing plan.
|
|
149
|
+
'tildes.net': {
|
|
150
|
+
search: {
|
|
151
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
152
|
+
itemSelector: 'article.topic',
|
|
153
|
+
fields: { title: 'h1.topic-title a', url: 'h1.topic-title a@href', group: 'a.link-group' },
|
|
154
|
+
},
|
|
155
|
+
},
|
|
156
|
+
'remoteok.com': {
|
|
157
|
+
// Filters live in the path rather than a query string.
|
|
158
|
+
search: {
|
|
159
|
+
url: (o, t) => `${o}/remote-${enc(t.input.query)}-jobs`,
|
|
160
|
+
itemSelector: 'tr.job',
|
|
161
|
+
fields: { id: '@data-slug', url: '@data-url' },
|
|
162
|
+
},
|
|
163
|
+
},
|
|
164
|
+
// Held-out 3. Authored from the rendered DOM only, against the browse axis
|
|
165
|
+
// each site allows, because all five disallow /search.
|
|
166
|
+
// A: client-heavy forum. /tag/{tag} redirects to /tag/{tag}/{id}.
|
|
167
|
+
'meta.discourse.org': {
|
|
168
|
+
search: {
|
|
169
|
+
url: (o, t) => `${o}/tag/${enc(t.input.query)}`,
|
|
170
|
+
itemSelector: 'tr.topic-list-item',
|
|
171
|
+
fields: { title: 'a.title', url: 'a.title@href' },
|
|
172
|
+
},
|
|
173
|
+
// ?page=N is the site's own pagination: the tag page emits it as
|
|
174
|
+
// link[rel=next]. It also reorders the first page, so page 1 here is not
|
|
175
|
+
// the listing the bare /tag/{tag} of `search` returns.
|
|
176
|
+
list: {
|
|
177
|
+
url: (o, t) => `${o}/tag/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
178
|
+
itemSelector: 'tr.topic-list-item',
|
|
179
|
+
fields: { title: 'a.title', url: 'a.title@href' },
|
|
180
|
+
},
|
|
181
|
+
// A bare topic id is enough; the site redirects to the slugged form.
|
|
182
|
+
detail: {
|
|
183
|
+
url: (o, t) => `${o}/t/${enc(t.input.id)}`,
|
|
184
|
+
itemSelector: '#topic-title',
|
|
185
|
+
fields: { title: 'h1 a', url: 'h1 a@href', category: 'span.category-name' },
|
|
186
|
+
},
|
|
187
|
+
},
|
|
188
|
+
// B: faceted catalog. The browse axis is the format, which is a path segment.
|
|
189
|
+
'loc.gov': {
|
|
190
|
+
search: {
|
|
191
|
+
url: (o, t) => `${o}/${enc(t.input.query)}/`,
|
|
192
|
+
itemSelector: 'li.item',
|
|
193
|
+
fields: { title: 'span.item-description-title', url: 'a@href' },
|
|
194
|
+
},
|
|
195
|
+
// Pagination is ?sp=N, not ?page=N — the listing's own next/prev links say so.
|
|
196
|
+
list: {
|
|
197
|
+
url: (o, t) => `${o}/${enc(t.input.query)}/?sp=${enc(t.input.page)}`,
|
|
198
|
+
itemSelector: 'li.item',
|
|
199
|
+
fields: { title: 'span.item-description-title', url: 'a@href' },
|
|
200
|
+
},
|
|
201
|
+
// No `url` field: an item page carries no link to itself inside its own
|
|
202
|
+
// root, so the detail tasks narrow the oracle to the fields that exist.
|
|
203
|
+
detail: {
|
|
204
|
+
url: (o, t) => `${o}/item/${enc(t.input.id)}/`,
|
|
205
|
+
itemSelector: '.item-container',
|
|
206
|
+
fields: { title: 'h1 cite', format: 'h1 a.format-label' },
|
|
207
|
+
},
|
|
208
|
+
},
|
|
209
|
+
// C: commercial catalog. /tag/{tag} redirects to /discover/{tag}. The
|
|
210
|
+
// `data-v-*` attributes are build hashes and must never enter a selector.
|
|
211
|
+
'bandcamp.com': {
|
|
212
|
+
search: {
|
|
213
|
+
url: (o, t) => `${o}/tag/${enc(t.input.query)}`,
|
|
214
|
+
itemSelector: 'li.card-item',
|
|
215
|
+
fields: { title: 'a.stretch-link', url: 'a.stretch-link@href' },
|
|
216
|
+
},
|
|
217
|
+
},
|
|
218
|
+
// D: the stateful-pagination slot.
|
|
219
|
+
'lemmy.world': {
|
|
220
|
+
// Slot D exists to put the recipe model's missing pagination state under
|
|
221
|
+
// load, so `page` is a second input rather than a fixed 1.
|
|
222
|
+
search: {
|
|
223
|
+
url: (o, t) => `${o}/c/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
224
|
+
itemSelector: 'article.post-container',
|
|
225
|
+
fields: { title: 'a.link-dark', url: 'a.link-dark@href' },
|
|
226
|
+
},
|
|
227
|
+
list: {
|
|
228
|
+
url: (o, t) => `${o}/c/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
229
|
+
itemSelector: 'article.post-container',
|
|
230
|
+
fields: { title: 'a.link-dark', url: 'a.link-dark@href' },
|
|
231
|
+
},
|
|
232
|
+
// .post-listing, not article.post-container: the latter is emitted twice
|
|
233
|
+
// per post, once for each of the narrow and wide layouts.
|
|
234
|
+
detail: {
|
|
235
|
+
url: (o, t) => `${o}/post/${enc(t.input.id)}`,
|
|
236
|
+
itemSelector: '.post-listing',
|
|
237
|
+
fields: { title: 'a.link-dark', url: 'a.link-dark@href', community: 'a.community-link' },
|
|
238
|
+
},
|
|
239
|
+
},
|
|
240
|
+
// E: positive control. Both inputs are visible in the URL, which is the
|
|
241
|
+
// property the control exists to verify the engine can still exploit.
|
|
242
|
+
'openlibrary.org': {
|
|
243
|
+
search: {
|
|
244
|
+
url: (o, t) => `${o}/trending/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
245
|
+
itemSelector: 'li.searchResultItem',
|
|
246
|
+
fields: {
|
|
247
|
+
title: 'h3.booktitle a.results',
|
|
248
|
+
url: 'h3.booktitle a.results@href',
|
|
249
|
+
author: 'span.bookauthor a',
|
|
250
|
+
},
|
|
251
|
+
},
|
|
252
|
+
list: {
|
|
253
|
+
url: (o, t) => `${o}/trending/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
254
|
+
itemSelector: 'li.searchResultItem',
|
|
255
|
+
fields: {
|
|
256
|
+
title: 'h3.booktitle a.results',
|
|
257
|
+
url: 'h3.booktitle a.results@href',
|
|
258
|
+
author: 'span.bookauthor a',
|
|
259
|
+
},
|
|
260
|
+
},
|
|
261
|
+
// The first /works/ link inside the panel is the work's own, carrying the
|
|
262
|
+
// edition the page chose to display — the same shape the listing yields.
|
|
263
|
+
detail: {
|
|
264
|
+
url: (o, t) => `${o}/works/${enc(t.input.id)}`,
|
|
265
|
+
itemSelector: 'div.workDetails',
|
|
266
|
+
fields: {
|
|
267
|
+
title: 'h1.work-title',
|
|
268
|
+
url: 'a[href^="/works/OL"]@href',
|
|
269
|
+
author: 'h2.edition-byline a',
|
|
270
|
+
},
|
|
271
|
+
},
|
|
272
|
+
},
|
|
273
|
+
// Held-out 4. Authored from the rendered DOM only, on the browse axis each
|
|
274
|
+
// site allows.
|
|
275
|
+
// A: cursor slot. The tag timeline renders its statuses lazily, so most of the
|
|
276
|
+
// 20 `article` elements are empty placeholders until they enter the viewport;
|
|
277
|
+
// the `:has` scopes the listing to the statuses that actually rendered.
|
|
278
|
+
'mastodon.social': {
|
|
279
|
+
search: {
|
|
280
|
+
url: (o, t) => `${o}/tags/${enc(t.input.query)}`,
|
|
281
|
+
itemSelector: 'article:has(a.status__relative-time)',
|
|
282
|
+
fields: { title: '.status__content', url: 'a.status__relative-time@href' },
|
|
283
|
+
},
|
|
284
|
+
// Identical to `search`: the timeline's only next-page affordance is a
|
|
285
|
+
// "Load more" button that changes no URL, so page 2 has no address to name.
|
|
286
|
+
list: {
|
|
287
|
+
url: (o, t) => `${o}/tags/${enc(t.input.query)}`,
|
|
288
|
+
itemSelector: 'article:has(a.status__relative-time)',
|
|
289
|
+
fields: { title: '.status__content', url: 'a.status__relative-time@href' },
|
|
290
|
+
},
|
|
291
|
+
},
|
|
292
|
+
// B: cursor slot. The grid carries no text at all — no caption, no alt, no
|
|
293
|
+
// link title — so `url` is the only field the surface affords.
|
|
294
|
+
'pixelfed.social': {
|
|
295
|
+
search: {
|
|
296
|
+
url: (o, t) => `${o}/discover/tags/${enc(t.input.query)}`,
|
|
297
|
+
itemSelector: '.hashtag-post-square',
|
|
298
|
+
fields: { url: 'a@href' },
|
|
299
|
+
},
|
|
300
|
+
// Identical to `search` for the same reason as mastodon: the page offers no
|
|
301
|
+
// next link, no page parameter, and scrolling adds nothing.
|
|
302
|
+
list: {
|
|
303
|
+
url: (o, t) => `${o}/discover/tags/${enc(t.input.query)}`,
|
|
304
|
+
itemSelector: '.hashtag-post-square',
|
|
305
|
+
fields: { url: 'a@href' },
|
|
306
|
+
},
|
|
307
|
+
},
|
|
308
|
+
// C: metadata catalog. /search, /tag/ and /recording/ are disallowed; the
|
|
309
|
+
// allowed listing is an artist's release-group table, so the axis is an MBID.
|
|
310
|
+
// `artist` is the composed credit column, which is what slot C exists to test.
|
|
311
|
+
'musicbrainz.org': {
|
|
312
|
+
search: {
|
|
313
|
+
url: (o, t) => `${o}/artist/${enc(t.input.query)}`,
|
|
314
|
+
itemSelector: 'tr.odd, tr.even',
|
|
315
|
+
fields: { title: 'a.wrap-anywhere', url: 'a.wrap-anywhere@href', artist: 'td:nth-child(3)' },
|
|
316
|
+
},
|
|
317
|
+
// ?page=N is the listing's own pagination: the table emits numbered links.
|
|
318
|
+
list: {
|
|
319
|
+
url: (o, t) => `${o}/artist/${enc(t.input.query)}?page=${enc(t.input.page)}`,
|
|
320
|
+
itemSelector: 'tr.odd, tr.even',
|
|
321
|
+
fields: { title: 'a.wrap-anywhere', url: 'a.wrap-anywhere@href', artist: 'td:nth-child(3)' },
|
|
322
|
+
},
|
|
323
|
+
detail: {
|
|
324
|
+
url: (o, t) => `${o}/release-group/${enc(t.input.id)}`,
|
|
325
|
+
itemSelector: 'h1',
|
|
326
|
+
fields: { title: 'a', url: 'a@href' },
|
|
327
|
+
},
|
|
328
|
+
},
|
|
329
|
+
// D: client-heavy forum. /search is disallowed; the tag feed is the axis.
|
|
330
|
+
'dev.to': {
|
|
331
|
+
search: {
|
|
332
|
+
url: (o, t) => `${o}/t/${enc(t.input.query)}`,
|
|
333
|
+
itemSelector: 'div.crayons-story',
|
|
334
|
+
fields: { title: 'h2.crayons-story__title a', url: 'h2.crayons-story__title a@href' },
|
|
335
|
+
},
|
|
336
|
+
// Pagination is a path segment, /t/{tag}/page/N, not a query parameter.
|
|
337
|
+
list: {
|
|
338
|
+
url: (o, t) => `${o}/t/${enc(t.input.query)}/page/${enc(t.input.page)}`,
|
|
339
|
+
itemSelector: 'div.crayons-story',
|
|
340
|
+
fields: { title: 'h2.crayons-story__title a', url: 'h2.crayons-story__title a@href' },
|
|
341
|
+
},
|
|
342
|
+
// An article id is a two-segment path, so each segment is encoded on its
|
|
343
|
+
// own; encoding the whole id would escape the separator.
|
|
344
|
+
// No `url` field: nothing inside the article root links to the article.
|
|
345
|
+
detail: {
|
|
346
|
+
url: (o, t) => `${o}/${String(t.input.id).split('/').map(enc).join('/')}`,
|
|
347
|
+
itemSelector: 'article.crayons-article',
|
|
348
|
+
fields: { title: 'h1', author: 'a.crayons-link' },
|
|
349
|
+
},
|
|
350
|
+
},
|
|
351
|
+
// E: positive control. Every class on this site is a build hash (db7ee1ac …),
|
|
352
|
+
// so selectors use the anchors, the landmark ids and the tag names instead.
|
|
353
|
+
'npmjs.com': {
|
|
354
|
+
search: {
|
|
355
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
356
|
+
itemSelector: 'a[href^="/package/"]',
|
|
357
|
+
fields: { title: '', url: '@href' },
|
|
358
|
+
},
|
|
359
|
+
// npm's page parameter is zero-based — its own "1" link is page=0 — so the
|
|
360
|
+
// task's `page` is that index, and perPage matches the emitted href.
|
|
361
|
+
list: {
|
|
362
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}&page=${enc(t.input.page)}&perPage=20`,
|
|
363
|
+
itemSelector: 'a[href^="/package/"]',
|
|
364
|
+
fields: { title: '', url: '@href' },
|
|
365
|
+
},
|
|
366
|
+
// The readme tab is the package's own canonical link, and the only one on
|
|
367
|
+
// the page that is not a build-hash class.
|
|
368
|
+
detail: {
|
|
369
|
+
url: (o, t) => `${o}/package/${enc(t.input.id)}`,
|
|
370
|
+
itemSelector: 'main#main',
|
|
371
|
+
fields: { title: 'h1', url: 'a#package-tab-readme@href' },
|
|
372
|
+
},
|
|
373
|
+
},
|
|
374
|
+
'itch.io': {
|
|
375
|
+
// /search is disallowed; the tag browse pages carry the same result shape.
|
|
376
|
+
search: {
|
|
377
|
+
url: (o, t) => `${o}/games/tag-${enc(t.input.query)}`,
|
|
378
|
+
itemSelector: '.game_cell',
|
|
379
|
+
fields: { id: '@data-game_id', title: 'a.title', url: 'a.title@href' },
|
|
380
|
+
},
|
|
381
|
+
},
|
|
382
|
+
'arbeitnow.com': {
|
|
383
|
+
search: {
|
|
384
|
+
url: (o, t) => `${o}/jobs?search=${enc(t.input.query)}`,
|
|
385
|
+
itemSelector: 'h3.flex.items-center',
|
|
386
|
+
fields: { title: 'a[href*="/jobs/"]', url: 'a[href*="/jobs/"]@href' },
|
|
387
|
+
},
|
|
388
|
+
},
|
|
389
|
+
// Known-hard case: results live inside app-root's shadow DOM, which neither
|
|
390
|
+
// querySelectorAll nor page.content() can reach. Expected to stay at L3, and
|
|
391
|
+
// included so that the boundary is a measured number rather than an opinion.
|
|
392
|
+
'archive.org': {
|
|
393
|
+
search: {
|
|
394
|
+
url: (o, t) => `${o}/search?query=${enc(t.input.query)}`,
|
|
395
|
+
itemSelector: 'a[href^="/details/"]',
|
|
396
|
+
fields: { url: '@href' },
|
|
397
|
+
},
|
|
398
|
+
},
|
|
399
|
+
// Harness check only. Same server as siteA, graded paired-live with strict
|
|
400
|
+
// ordering, which is the only way orderingAgreement runs outside unit tests:
|
|
401
|
+
// every wild site reorders equally-ranked results, and the other fixtures are
|
|
402
|
+
// graded golden.
|
|
403
|
+
siteOrdered: {
|
|
404
|
+
search: {
|
|
405
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
406
|
+
itemSelector: 'li.result',
|
|
407
|
+
fields: { id: '@data-id', title: 'a.title' },
|
|
408
|
+
},
|
|
409
|
+
},
|
|
410
|
+
// Harness check only. The selector is meant to find nothing, so the browser
|
|
411
|
+
// returns an unusable baseline and the exclusion path fires deterministically
|
|
412
|
+
// instead of waiting for a real site to time out.
|
|
413
|
+
siteBroken: {
|
|
414
|
+
search: {
|
|
415
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
416
|
+
itemSelector: '.deliberately-absent',
|
|
417
|
+
fields: { title: 'a.title' },
|
|
418
|
+
},
|
|
419
|
+
},
|
|
420
|
+
// Harness check only. The server serves siteA's page to a browser and refuses
|
|
421
|
+
// the engine's client, so the browser plan is siteA's plan exactly.
|
|
422
|
+
siteRefusing: SITE_A_PLANS,
|
|
423
|
+
// Harness check only. Also siteA's page to a browser, but the engine's client
|
|
424
|
+
// gets a 200 challenge page instead of a 403, so the browser plan has to be
|
|
425
|
+
// siteA's exactly: a re-record must compile the recipe the site already has.
|
|
426
|
+
siteStub: SITE_A_PLANS,
|
|
427
|
+
// Harness check only. The page renders `title by (editor ?? author)`, so the
|
|
428
|
+
// title can only be reproduced by a template with a coalesce in it.
|
|
429
|
+
siteCoalesce: {
|
|
430
|
+
search: {
|
|
431
|
+
url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
|
|
432
|
+
itemSelector: 'li.result',
|
|
433
|
+
fields: { id: '@data-id', title: 'a', url: 'a@href' },
|
|
434
|
+
},
|
|
435
|
+
},
|
|
436
|
+
};
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { startFixture, sendHtml } from './harness.js';
|
|
2
|
+
import { isBrowser } from './refusing.js';
|
|
3
|
+
import { page } from './ssr.js';
|
|
4
|
+
import { DATASET, search } from './data.js';
|
|
5
|
+
/**
|
|
6
|
+
* Filters honestly for everyone and answers the engine's own client with a
|
|
7
|
+
* neighbouring record, the way a site that serves non-browser clients a
|
|
8
|
+
* different rendering does.
|
|
9
|
+
*
|
|
10
|
+
* Split on client hints, which a browser sends and `fetch` does not — the same
|
|
11
|
+
* discriminator `refusing.ts` already uses, and a real difference between the
|
|
12
|
+
* two transports rather than a count of how many requests have arrived.
|
|
13
|
+
*
|
|
14
|
+
* Everything the probes can see from HTTP alone holds: the answer moves with
|
|
15
|
+
* the query, it is the same twice running, and a query nothing matches comes
|
|
16
|
+
* back empty. Only asking the browser the same untaught question shows the two
|
|
17
|
+
* paths do not agree, which is the one signal this fixture exists to fail.
|
|
18
|
+
*/
|
|
19
|
+
export function startCloakingFixture() {
|
|
20
|
+
return startFixture('siteCloaking', (req, res) => {
|
|
21
|
+
const url = new URL(req.url ?? '/', 'http://localhost');
|
|
22
|
+
if (url.pathname !== '/search') {
|
|
23
|
+
sendHtml(res, page('Home', '<a href="/search?q=rust">search</a>'));
|
|
24
|
+
return;
|
|
25
|
+
}
|
|
26
|
+
const hits = search(url.searchParams.get('q') ?? '', Number(url.searchParams.get('page') ?? '1'));
|
|
27
|
+
const shown = isBrowser(req)
|
|
28
|
+
? hits
|
|
29
|
+
: hits.map((hit) => DATASET[(DATASET.indexOf(hit) + 1) % DATASET.length]);
|
|
30
|
+
const rows = shown
|
|
31
|
+
.map((r) => `<li class="result" data-id="${r.id}">
|
|
32
|
+
<a class="title" href="/item/${r.id}">${r.title}</a>
|
|
33
|
+
</li>`)
|
|
34
|
+
.join('');
|
|
35
|
+
sendHtml(res, page('Search', `<ul id="results">${rows}</ul>`));
|
|
36
|
+
});
|
|
37
|
+
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import { startFixture, sendHtml, sendJson, sendJs } from './harness.js';
|
|
2
|
+
import { search, DATASET } from './data.js';
|
|
3
|
+
const EDITORS = new Map(DATASET.map((r, i) => [r.id, i % 3 === 1 ? `editor-${i % 5}` : null]));
|
|
4
|
+
function withEditor(r) {
|
|
5
|
+
return { ...r, editor: EDITORS.get(r.id) ?? null };
|
|
6
|
+
}
|
|
7
|
+
const BOOT_SCRIPT = `
|
|
8
|
+
const params = new URLSearchParams(location.search)
|
|
9
|
+
const q = params.get('q') || ''
|
|
10
|
+
const page = params.get('page') || '1'
|
|
11
|
+
async function boot() {
|
|
12
|
+
const res = await fetch('/api/search?q=' + encodeURIComponent(q) + '&page=' + encodeURIComponent(page))
|
|
13
|
+
const data = await res.json()
|
|
14
|
+
const ul = document.createElement('ul')
|
|
15
|
+
ul.id = 'results'
|
|
16
|
+
for (const r of data.results) {
|
|
17
|
+
const li = document.createElement('li')
|
|
18
|
+
li.className = 'result'
|
|
19
|
+
li.dataset.id = r.id
|
|
20
|
+
const a = document.createElement('a')
|
|
21
|
+
a.href = '/item/' + r.id
|
|
22
|
+
a.textContent = r.title + ' by ' + (r.editor ?? r.author)
|
|
23
|
+
li.appendChild(a)
|
|
24
|
+
ul.appendChild(li)
|
|
25
|
+
}
|
|
26
|
+
document.getElementById('app').appendChild(ul)
|
|
27
|
+
}
|
|
28
|
+
boot()
|
|
29
|
+
`;
|
|
30
|
+
export function startCoalesceFixture() {
|
|
31
|
+
return startFixture('siteCoalesce', (req, res) => {
|
|
32
|
+
const url = new URL(req.url ?? '/', 'http://localhost');
|
|
33
|
+
switch (true) {
|
|
34
|
+
case url.pathname === '/search':
|
|
35
|
+
sendHtml(res, `<!doctype html><html><head><title>Search</title>
|
|
36
|
+
<script src="/app.js" defer></script></head>
|
|
37
|
+
<body><header>Site Coalesce</header><div id="app"></div></body></html>`);
|
|
38
|
+
return;
|
|
39
|
+
case url.pathname === '/app.js':
|
|
40
|
+
sendJs(res, BOOT_SCRIPT);
|
|
41
|
+
return;
|
|
42
|
+
case url.pathname === '/api/search': {
|
|
43
|
+
const q = url.searchParams.get('q') ?? '';
|
|
44
|
+
const page = Number(url.searchParams.get('page') ?? '1');
|
|
45
|
+
sendJson(res, { query: q, page, total: DATASET.length, results: search(q, page).map(withEditor) });
|
|
46
|
+
return;
|
|
47
|
+
}
|
|
48
|
+
default:
|
|
49
|
+
sendJson(res, { error: 'not found' }, 404);
|
|
50
|
+
}
|
|
51
|
+
});
|
|
52
|
+
}
|