webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,129 @@
1
+ /**
2
+ * Per-site oracle defaults. A site absent here is graded `golden`, which is
3
+ * what every run before this contract used.
4
+ *
5
+ * `volatility` is documentation: it records why a mode was chosen, so a later
6
+ * reader can tell a considered choice from a convenient one.
7
+ */
8
+ export const ORACLES = {
9
+ // Job boards: the whole listing turns over within minutes, so a stored
10
+ // capture measures its own age rather than the engine.
11
+ 'remoteok.com': {
12
+ mode: 'paired-live',
13
+ volatility: 'high',
14
+ compare: { ordering: 'ignore', fields: ['id', 'url'] },
15
+ },
16
+ 'arbeitnow.com': {
17
+ mode: 'paired-live',
18
+ volatility: 'high',
19
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
20
+ // The one site measured so far that can silently ignore its own search
21
+ // input: the parameter is discarded on redirect and every query returns the
22
+ // same unfiltered front page. Declared here and nowhere else, because a
23
+ // rule that fires on sites it was not designed for is an oracle bug.
24
+ semantics: { input: 'query', fields: ['title'], match: 'contains-token', minShare: 0.5 },
25
+ },
26
+ // A marketplace browse page and a small forum: the set moves over days, and
27
+ // both reorder equally-ranked items between requests.
28
+ 'itch.io': {
29
+ mode: 'paired-live',
30
+ volatility: 'medium',
31
+ compare: { ordering: 'ignore', fields: ['id', 'title', 'url'] },
32
+ },
33
+ 'tildes.net': {
34
+ mode: 'paired-live',
35
+ volatility: 'medium',
36
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
37
+ },
38
+ // Registries and documentation: a version bump is the only churn, and result
39
+ // ranking is stable enough to hold to.
40
+ 'docs.rs': { mode: 'golden', volatility: 'low' },
41
+ 'crates.io': { mode: 'golden', volatility: 'low' },
42
+ 'hex.pm': { mode: 'golden', volatility: 'low' },
43
+ 'jsr.io': { mode: 'golden', volatility: 'low' },
44
+ 'pkg.go.dev': {
45
+ mode: 'golden',
46
+ volatility: 'low',
47
+ // Ranks ties differently between requests; the set is stable, the order is not.
48
+ compare: { ordering: 'ignore' },
49
+ },
50
+ 'hn.algolia.com': { mode: 'golden', volatility: 'low' },
51
+ 'flathub.org': {
52
+ mode: 'golden',
53
+ volatility: 'medium',
54
+ compare: { ordering: 'ignore' },
55
+ },
56
+ // Held-out 3. Four of the five browse axes turn over on their own schedule —
57
+ // a forum's tag feed, a federated community, a music catalog's discover
58
+ // listing, and a trending chart are all volatile by construction — so they
59
+ // are graded against the adjacent browser run rather than a stored capture.
60
+ 'meta.discourse.org': {
61
+ mode: 'paired-live', volatility: 'high',
62
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
63
+ },
64
+ 'lemmy.world': {
65
+ mode: 'paired-live', volatility: 'high',
66
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
67
+ },
68
+ 'bandcamp.com': {
69
+ mode: 'paired-live', volatility: 'high',
70
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
71
+ },
72
+ 'openlibrary.org': {
73
+ mode: 'paired-live', volatility: 'high',
74
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
75
+ },
76
+ // The one held-out 3 axis that does not move: a format listing is a catalog,
77
+ // not a feed.
78
+ 'loc.gov': {
79
+ mode: 'golden', volatility: 'low',
80
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
81
+ },
82
+ // Held-out 4. The three feeds turn over continuously — two fediverse tag
83
+ // timelines and a forum's tag feed — so they are graded against the adjacent
84
+ // browser run. pixelfed's grid carries no text of any kind, so `url` is the
85
+ // only field there is to compare.
86
+ 'mastodon.social': {
87
+ mode: 'paired-live', volatility: 'high',
88
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
89
+ },
90
+ 'pixelfed.social': {
91
+ mode: 'paired-live', volatility: 'high',
92
+ compare: { ordering: 'ignore', fields: ['url'] },
93
+ },
94
+ 'dev.to': {
95
+ mode: 'paired-live', volatility: 'high',
96
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
97
+ },
98
+ // A catalog and a registry: a discography and a search ranking both hold
99
+ // still long enough for a stored capture to be the scorer. `artist` is graded
100
+ // on musicbrainz because the composed credit column is what slot C measures.
101
+ 'musicbrainz.org': {
102
+ mode: 'golden', volatility: 'low',
103
+ compare: { ordering: 'ignore', fields: ['title', 'url', 'artist'] },
104
+ },
105
+ 'npmjs.com': {
106
+ mode: 'golden', volatility: 'low',
107
+ compare: { ordering: 'ignore', fields: ['title', 'url'] },
108
+ },
109
+ // Fixtures serve a fixed dataset in a fixed order, which makes them the only
110
+ // sites where ordering is answerable at all — every wild site measured so far
111
+ // reorders equally-ranked results between requests.
112
+ siteA: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
113
+ siteB: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
114
+ siteC: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
115
+ siteCoalesce: { mode: 'golden', volatility: 'low', compare: { ordering: 'strict' } },
116
+ // Paired-live so that an unusable baseline reaches baselineValidity; under
117
+ // golden the stored capture is the scorer and never consults the browser.
118
+ siteBroken: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
119
+ // The one place orderingAgreement is reachable: paired-live plus a stable order.
120
+ siteOrdered: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'strict' } },
121
+ // Paired-live because the whole point is that the browser side stays healthy
122
+ // while the engine's client is refused; a stored golden never consults the
123
+ // browser, so it could not show the difference.
124
+ siteRefusing: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
125
+ // Paired-live for siteRefusing's reason, and because the browser side is the
126
+ // half that stays healthy here: it is the engine's client alone that is sent
127
+ // a different page.
128
+ siteStub: { mode: 'paired-live', volatility: 'low', compare: { ordering: 'ignore' } },
129
+ };
@@ -0,0 +1,436 @@
1
+ const enc = (v) => encodeURIComponent(String(v));
2
+ /** Hoisted because siteRefusing serves this exact page; the two must not drift. */
3
+ const SITE_A_PLANS = {
4
+ search: {
5
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
6
+ itemSelector: 'li.result',
7
+ fields: { id: '@data-id', title: 'a.title', url: 'a.title@href' },
8
+ },
9
+ list: {
10
+ url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
11
+ itemSelector: 'li.result',
12
+ fields: { id: '@data-id', title: 'a.title', url: 'a.title@href' },
13
+ },
14
+ detail: {
15
+ url: (o, t) => `${o}/item/${enc(t.input.id)}`,
16
+ itemSelector: 'article#detail',
17
+ fields: { id: '@data-id', title: 'h1.title', author: 'span.author' },
18
+ },
19
+ };
20
+ /**
21
+ * Per-site knowledge of where to go and what to read — exactly what a learned
22
+ * recipe replaces. Fixture plans are exact; wild plans are verified against the
23
+ * live pages before any benchmark number is trusted.
24
+ */
25
+ export const PLANS = {
26
+ siteA: SITE_A_PLANS,
27
+ // siteB's API returns no url; the page composes it from the id, so a recipe
28
+ // can only match the browser by learning that derivation.
29
+ siteB: {
30
+ search: {
31
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
32
+ itemSelector: 'li.result',
33
+ fields: { id: '@data-id', title: '', url: 'a.permalink@href' },
34
+ },
35
+ list: {
36
+ url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
37
+ itemSelector: 'li.result',
38
+ fields: { id: '@data-id', title: '', url: 'a.permalink@href' },
39
+ },
40
+ detail: {
41
+ url: (o, t) => `${o}/item/${enc(t.input.id)}`,
42
+ itemSelector: 'article#detail',
43
+ fields: { id: '@data-id', title: 'h1.title', author: 'span.author' },
44
+ },
45
+ },
46
+ siteC: {
47
+ search: {
48
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
49
+ itemSelector: 'li.result',
50
+ fields: { id: '@data-id', title: '' },
51
+ },
52
+ list: {
53
+ url: (o, t) => `${o}/search?q=&page=${enc(t.input.page)}`,
54
+ itemSelector: 'li.result',
55
+ fields: { id: '@data-id', title: '' },
56
+ },
57
+ detail: {
58
+ url: (o, t) => `${o}/item/${enc(t.input.id)}`,
59
+ itemSelector: 'article#detail',
60
+ fields: { id: '@data-id', title: '' },
61
+ },
62
+ },
63
+ // crates.io is a Svelte app; its generated class names (svelte-10p0ryf) change
64
+ // on every build, so only the semantic ones are usable as selectors.
65
+ 'crates.io': {
66
+ search: {
67
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
68
+ itemSelector: '.crate-row',
69
+ fields: { title: 'a.name', url: 'a.name@href' },
70
+ },
71
+ detail: {
72
+ url: (o, t) => `${o}/crates/${enc(t.input.id)}`,
73
+ itemSelector: 'h1.heading',
74
+ fields: { title: '' },
75
+ },
76
+ },
77
+ 'pkg.go.dev': {
78
+ search: {
79
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
80
+ itemSelector: '.SearchSnippet',
81
+ fields: { title: 'h2 a', url: 'h2 a@href' },
82
+ },
83
+ detail: {
84
+ url: (o, t) => `${o}/${String(t.input.id)}`,
85
+ itemSelector: '.UnitHeader-titleHeading',
86
+ fields: { title: '' },
87
+ },
88
+ },
89
+ 'hn.algolia.com': {
90
+ search: {
91
+ url: (o, t) => `${o}/?query=${enc(t.input.query)}`,
92
+ itemSelector: '.Story',
93
+ fields: { title: '.Story_title a', url: '.Story_title a@href' },
94
+ },
95
+ },
96
+ // ---------------------------------------------------------------------------
97
+ // Held-out sites. Selectors were authored from the rendered DOM, as any user
98
+ // of this tool would; their API payloads were deliberately not inspected.
99
+ // ---------------------------------------------------------------------------
100
+ 'hex.pm': {
101
+ search: {
102
+ url: (o, t) => `${o}/packages?search=${enc(t.input.query)}`,
103
+ itemSelector: 'li:has(a[href^="/packages/"])',
104
+ fields: { title: 'a[href^="/packages/"]', url: 'a[href^="/packages/"]@href' },
105
+ },
106
+ detail: {
107
+ url: (o, t) => `${o}/packages/${enc(t.input.id)}`,
108
+ itemSelector: 'h1',
109
+ fields: { title: '' },
110
+ },
111
+ },
112
+ 'docs.rs': {
113
+ search: {
114
+ url: (o, t) => `${o}/releases/search?query=${enc(t.input.query)}`,
115
+ itemSelector: 'a.release',
116
+ fields: { title: '.name', description: '.description', url: '@href' },
117
+ },
118
+ },
119
+ 'flathub.org': {
120
+ search: {
121
+ url: (o, t) => `${o}/apps/search?q=${enc(t.input.query)}`,
122
+ // Scoped to main: the same anchor shape appears in the nav and the
123
+ // recommendation strip, which would pad the result set with five rows
124
+ // the search API never returned.
125
+ itemSelector: 'main a[href^="/en/apps/"]',
126
+ fields: { title: 'span.truncate', url: '@href' },
127
+ },
128
+ detail: {
129
+ url: (o, t) => `${o}/apps/${String(t.input.id)}`,
130
+ itemSelector: 'h1',
131
+ fields: { title: '' },
132
+ },
133
+ },
134
+ 'jsr.io': {
135
+ search: {
136
+ url: (o, t) => `${o}/packages?search=${enc(t.input.query)}`,
137
+ itemSelector: 'li:has(a[href^="/@"])',
138
+ fields: { title: 'a[href^="/@"]', url: 'a[href^="/@"]@href' },
139
+ },
140
+ },
141
+ // ---------------------------------------------------------------------------
142
+ // Held-out 2. Same rule as before: selectors authored from the rendered DOM,
143
+ // payloads never inspected.
144
+ // ---------------------------------------------------------------------------
145
+ // No list intent for these three: none of them paginates through a URL
146
+ // parameter. tildes uses an after= cursor, itch.io and arbeitnow ignore
147
+ // ?page= entirely, and remoteok scrolls. A recipe replays one static request,
148
+ // so a second page is not expressible — a design limit, not a missing plan.
149
+ 'tildes.net': {
150
+ search: {
151
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
152
+ itemSelector: 'article.topic',
153
+ fields: { title: 'h1.topic-title a', url: 'h1.topic-title a@href', group: 'a.link-group' },
154
+ },
155
+ },
156
+ 'remoteok.com': {
157
+ // Filters live in the path rather than a query string.
158
+ search: {
159
+ url: (o, t) => `${o}/remote-${enc(t.input.query)}-jobs`,
160
+ itemSelector: 'tr.job',
161
+ fields: { id: '@data-slug', url: '@data-url' },
162
+ },
163
+ },
164
+ // Held-out 3. Authored from the rendered DOM only, against the browse axis
165
+ // each site allows, because all five disallow /search.
166
+ // A: client-heavy forum. /tag/{tag} redirects to /tag/{tag}/{id}.
167
+ 'meta.discourse.org': {
168
+ search: {
169
+ url: (o, t) => `${o}/tag/${enc(t.input.query)}`,
170
+ itemSelector: 'tr.topic-list-item',
171
+ fields: { title: 'a.title', url: 'a.title@href' },
172
+ },
173
+ // ?page=N is the site's own pagination: the tag page emits it as
174
+ // link[rel=next]. It also reorders the first page, so page 1 here is not
175
+ // the listing the bare /tag/{tag} of `search` returns.
176
+ list: {
177
+ url: (o, t) => `${o}/tag/${enc(t.input.query)}?page=${enc(t.input.page)}`,
178
+ itemSelector: 'tr.topic-list-item',
179
+ fields: { title: 'a.title', url: 'a.title@href' },
180
+ },
181
+ // A bare topic id is enough; the site redirects to the slugged form.
182
+ detail: {
183
+ url: (o, t) => `${o}/t/${enc(t.input.id)}`,
184
+ itemSelector: '#topic-title',
185
+ fields: { title: 'h1 a', url: 'h1 a@href', category: 'span.category-name' },
186
+ },
187
+ },
188
+ // B: faceted catalog. The browse axis is the format, which is a path segment.
189
+ 'loc.gov': {
190
+ search: {
191
+ url: (o, t) => `${o}/${enc(t.input.query)}/`,
192
+ itemSelector: 'li.item',
193
+ fields: { title: 'span.item-description-title', url: 'a@href' },
194
+ },
195
+ // Pagination is ?sp=N, not ?page=N — the listing's own next/prev links say so.
196
+ list: {
197
+ url: (o, t) => `${o}/${enc(t.input.query)}/?sp=${enc(t.input.page)}`,
198
+ itemSelector: 'li.item',
199
+ fields: { title: 'span.item-description-title', url: 'a@href' },
200
+ },
201
+ // No `url` field: an item page carries no link to itself inside its own
202
+ // root, so the detail tasks narrow the oracle to the fields that exist.
203
+ detail: {
204
+ url: (o, t) => `${o}/item/${enc(t.input.id)}/`,
205
+ itemSelector: '.item-container',
206
+ fields: { title: 'h1 cite', format: 'h1 a.format-label' },
207
+ },
208
+ },
209
+ // C: commercial catalog. /tag/{tag} redirects to /discover/{tag}. The
210
+ // `data-v-*` attributes are build hashes and must never enter a selector.
211
+ 'bandcamp.com': {
212
+ search: {
213
+ url: (o, t) => `${o}/tag/${enc(t.input.query)}`,
214
+ itemSelector: 'li.card-item',
215
+ fields: { title: 'a.stretch-link', url: 'a.stretch-link@href' },
216
+ },
217
+ },
218
+ // D: the stateful-pagination slot.
219
+ 'lemmy.world': {
220
+ // Slot D exists to put the recipe model's missing pagination state under
221
+ // load, so `page` is a second input rather than a fixed 1.
222
+ search: {
223
+ url: (o, t) => `${o}/c/${enc(t.input.query)}?page=${enc(t.input.page)}`,
224
+ itemSelector: 'article.post-container',
225
+ fields: { title: 'a.link-dark', url: 'a.link-dark@href' },
226
+ },
227
+ list: {
228
+ url: (o, t) => `${o}/c/${enc(t.input.query)}?page=${enc(t.input.page)}`,
229
+ itemSelector: 'article.post-container',
230
+ fields: { title: 'a.link-dark', url: 'a.link-dark@href' },
231
+ },
232
+ // .post-listing, not article.post-container: the latter is emitted twice
233
+ // per post, once for each of the narrow and wide layouts.
234
+ detail: {
235
+ url: (o, t) => `${o}/post/${enc(t.input.id)}`,
236
+ itemSelector: '.post-listing',
237
+ fields: { title: 'a.link-dark', url: 'a.link-dark@href', community: 'a.community-link' },
238
+ },
239
+ },
240
+ // E: positive control. Both inputs are visible in the URL, which is the
241
+ // property the control exists to verify the engine can still exploit.
242
+ 'openlibrary.org': {
243
+ search: {
244
+ url: (o, t) => `${o}/trending/${enc(t.input.query)}?page=${enc(t.input.page)}`,
245
+ itemSelector: 'li.searchResultItem',
246
+ fields: {
247
+ title: 'h3.booktitle a.results',
248
+ url: 'h3.booktitle a.results@href',
249
+ author: 'span.bookauthor a',
250
+ },
251
+ },
252
+ list: {
253
+ url: (o, t) => `${o}/trending/${enc(t.input.query)}?page=${enc(t.input.page)}`,
254
+ itemSelector: 'li.searchResultItem',
255
+ fields: {
256
+ title: 'h3.booktitle a.results',
257
+ url: 'h3.booktitle a.results@href',
258
+ author: 'span.bookauthor a',
259
+ },
260
+ },
261
+ // The first /works/ link inside the panel is the work's own, carrying the
262
+ // edition the page chose to display — the same shape the listing yields.
263
+ detail: {
264
+ url: (o, t) => `${o}/works/${enc(t.input.id)}`,
265
+ itemSelector: 'div.workDetails',
266
+ fields: {
267
+ title: 'h1.work-title',
268
+ url: 'a[href^="/works/OL"]@href',
269
+ author: 'h2.edition-byline a',
270
+ },
271
+ },
272
+ },
273
+ // Held-out 4. Authored from the rendered DOM only, on the browse axis each
274
+ // site allows.
275
+ // A: cursor slot. The tag timeline renders its statuses lazily, so most of the
276
+ // 20 `article` elements are empty placeholders until they enter the viewport;
277
+ // the `:has` scopes the listing to the statuses that actually rendered.
278
+ 'mastodon.social': {
279
+ search: {
280
+ url: (o, t) => `${o}/tags/${enc(t.input.query)}`,
281
+ itemSelector: 'article:has(a.status__relative-time)',
282
+ fields: { title: '.status__content', url: 'a.status__relative-time@href' },
283
+ },
284
+ // Identical to `search`: the timeline's only next-page affordance is a
285
+ // "Load more" button that changes no URL, so page 2 has no address to name.
286
+ list: {
287
+ url: (o, t) => `${o}/tags/${enc(t.input.query)}`,
288
+ itemSelector: 'article:has(a.status__relative-time)',
289
+ fields: { title: '.status__content', url: 'a.status__relative-time@href' },
290
+ },
291
+ },
292
+ // B: cursor slot. The grid carries no text at all — no caption, no alt, no
293
+ // link title — so `url` is the only field the surface affords.
294
+ 'pixelfed.social': {
295
+ search: {
296
+ url: (o, t) => `${o}/discover/tags/${enc(t.input.query)}`,
297
+ itemSelector: '.hashtag-post-square',
298
+ fields: { url: 'a@href' },
299
+ },
300
+ // Identical to `search` for the same reason as mastodon: the page offers no
301
+ // next link, no page parameter, and scrolling adds nothing.
302
+ list: {
303
+ url: (o, t) => `${o}/discover/tags/${enc(t.input.query)}`,
304
+ itemSelector: '.hashtag-post-square',
305
+ fields: { url: 'a@href' },
306
+ },
307
+ },
308
+ // C: metadata catalog. /search, /tag/ and /recording/ are disallowed; the
309
+ // allowed listing is an artist's release-group table, so the axis is an MBID.
310
+ // `artist` is the composed credit column, which is what slot C exists to test.
311
+ 'musicbrainz.org': {
312
+ search: {
313
+ url: (o, t) => `${o}/artist/${enc(t.input.query)}`,
314
+ itemSelector: 'tr.odd, tr.even',
315
+ fields: { title: 'a.wrap-anywhere', url: 'a.wrap-anywhere@href', artist: 'td:nth-child(3)' },
316
+ },
317
+ // ?page=N is the listing's own pagination: the table emits numbered links.
318
+ list: {
319
+ url: (o, t) => `${o}/artist/${enc(t.input.query)}?page=${enc(t.input.page)}`,
320
+ itemSelector: 'tr.odd, tr.even',
321
+ fields: { title: 'a.wrap-anywhere', url: 'a.wrap-anywhere@href', artist: 'td:nth-child(3)' },
322
+ },
323
+ detail: {
324
+ url: (o, t) => `${o}/release-group/${enc(t.input.id)}`,
325
+ itemSelector: 'h1',
326
+ fields: { title: 'a', url: 'a@href' },
327
+ },
328
+ },
329
+ // D: client-heavy forum. /search is disallowed; the tag feed is the axis.
330
+ 'dev.to': {
331
+ search: {
332
+ url: (o, t) => `${o}/t/${enc(t.input.query)}`,
333
+ itemSelector: 'div.crayons-story',
334
+ fields: { title: 'h2.crayons-story__title a', url: 'h2.crayons-story__title a@href' },
335
+ },
336
+ // Pagination is a path segment, /t/{tag}/page/N, not a query parameter.
337
+ list: {
338
+ url: (o, t) => `${o}/t/${enc(t.input.query)}/page/${enc(t.input.page)}`,
339
+ itemSelector: 'div.crayons-story',
340
+ fields: { title: 'h2.crayons-story__title a', url: 'h2.crayons-story__title a@href' },
341
+ },
342
+ // An article id is a two-segment path, so each segment is encoded on its
343
+ // own; encoding the whole id would escape the separator.
344
+ // No `url` field: nothing inside the article root links to the article.
345
+ detail: {
346
+ url: (o, t) => `${o}/${String(t.input.id).split('/').map(enc).join('/')}`,
347
+ itemSelector: 'article.crayons-article',
348
+ fields: { title: 'h1', author: 'a.crayons-link' },
349
+ },
350
+ },
351
+ // E: positive control. Every class on this site is a build hash (db7ee1ac …),
352
+ // so selectors use the anchors, the landmark ids and the tag names instead.
353
+ 'npmjs.com': {
354
+ search: {
355
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
356
+ itemSelector: 'a[href^="/package/"]',
357
+ fields: { title: '', url: '@href' },
358
+ },
359
+ // npm's page parameter is zero-based — its own "1" link is page=0 — so the
360
+ // task's `page` is that index, and perPage matches the emitted href.
361
+ list: {
362
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}&page=${enc(t.input.page)}&perPage=20`,
363
+ itemSelector: 'a[href^="/package/"]',
364
+ fields: { title: '', url: '@href' },
365
+ },
366
+ // The readme tab is the package's own canonical link, and the only one on
367
+ // the page that is not a build-hash class.
368
+ detail: {
369
+ url: (o, t) => `${o}/package/${enc(t.input.id)}`,
370
+ itemSelector: 'main#main',
371
+ fields: { title: 'h1', url: 'a#package-tab-readme@href' },
372
+ },
373
+ },
374
+ 'itch.io': {
375
+ // /search is disallowed; the tag browse pages carry the same result shape.
376
+ search: {
377
+ url: (o, t) => `${o}/games/tag-${enc(t.input.query)}`,
378
+ itemSelector: '.game_cell',
379
+ fields: { id: '@data-game_id', title: 'a.title', url: 'a.title@href' },
380
+ },
381
+ },
382
+ 'arbeitnow.com': {
383
+ search: {
384
+ url: (o, t) => `${o}/jobs?search=${enc(t.input.query)}`,
385
+ itemSelector: 'h3.flex.items-center',
386
+ fields: { title: 'a[href*="/jobs/"]', url: 'a[href*="/jobs/"]@href' },
387
+ },
388
+ },
389
+ // Known-hard case: results live inside app-root's shadow DOM, which neither
390
+ // querySelectorAll nor page.content() can reach. Expected to stay at L3, and
391
+ // included so that the boundary is a measured number rather than an opinion.
392
+ 'archive.org': {
393
+ search: {
394
+ url: (o, t) => `${o}/search?query=${enc(t.input.query)}`,
395
+ itemSelector: 'a[href^="/details/"]',
396
+ fields: { url: '@href' },
397
+ },
398
+ },
399
+ // Harness check only. Same server as siteA, graded paired-live with strict
400
+ // ordering, which is the only way orderingAgreement runs outside unit tests:
401
+ // every wild site reorders equally-ranked results, and the other fixtures are
402
+ // graded golden.
403
+ siteOrdered: {
404
+ search: {
405
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
406
+ itemSelector: 'li.result',
407
+ fields: { id: '@data-id', title: 'a.title' },
408
+ },
409
+ },
410
+ // Harness check only. The selector is meant to find nothing, so the browser
411
+ // returns an unusable baseline and the exclusion path fires deterministically
412
+ // instead of waiting for a real site to time out.
413
+ siteBroken: {
414
+ search: {
415
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
416
+ itemSelector: '.deliberately-absent',
417
+ fields: { title: 'a.title' },
418
+ },
419
+ },
420
+ // Harness check only. The server serves siteA's page to a browser and refuses
421
+ // the engine's client, so the browser plan is siteA's plan exactly.
422
+ siteRefusing: SITE_A_PLANS,
423
+ // Harness check only. Also siteA's page to a browser, but the engine's client
424
+ // gets a 200 challenge page instead of a 403, so the browser plan has to be
425
+ // siteA's exactly: a re-record must compile the recipe the site already has.
426
+ siteStub: SITE_A_PLANS,
427
+ // Harness check only. The page renders `title by (editor ?? author)`, so the
428
+ // title can only be reproduced by a template with a coalesce in it.
429
+ siteCoalesce: {
430
+ search: {
431
+ url: (o, t) => `${o}/search?q=${enc(t.input.query)}`,
432
+ itemSelector: 'li.result',
433
+ fields: { id: '@data-id', title: 'a', url: 'a@href' },
434
+ },
435
+ },
436
+ };
@@ -0,0 +1,37 @@
1
+ import { startFixture, sendHtml } from './harness.js';
2
+ import { isBrowser } from './refusing.js';
3
+ import { page } from './ssr.js';
4
+ import { DATASET, search } from './data.js';
5
+ /**
6
+ * Filters honestly for everyone and answers the engine's own client with a
7
+ * neighbouring record, the way a site that serves non-browser clients a
8
+ * different rendering does.
9
+ *
10
+ * Split on client hints, which a browser sends and `fetch` does not — the same
11
+ * discriminator `refusing.ts` already uses, and a real difference between the
12
+ * two transports rather than a count of how many requests have arrived.
13
+ *
14
+ * Everything the probes can see from HTTP alone holds: the answer moves with
15
+ * the query, it is the same twice running, and a query nothing matches comes
16
+ * back empty. Only asking the browser the same untaught question shows the two
17
+ * paths do not agree, which is the one signal this fixture exists to fail.
18
+ */
19
+ export function startCloakingFixture() {
20
+ return startFixture('siteCloaking', (req, res) => {
21
+ const url = new URL(req.url ?? '/', 'http://localhost');
22
+ if (url.pathname !== '/search') {
23
+ sendHtml(res, page('Home', '<a href="/search?q=rust">search</a>'));
24
+ return;
25
+ }
26
+ const hits = search(url.searchParams.get('q') ?? '', Number(url.searchParams.get('page') ?? '1'));
27
+ const shown = isBrowser(req)
28
+ ? hits
29
+ : hits.map((hit) => DATASET[(DATASET.indexOf(hit) + 1) % DATASET.length]);
30
+ const rows = shown
31
+ .map((r) => `<li class="result" data-id="${r.id}">
32
+ <a class="title" href="/item/${r.id}">${r.title}</a>
33
+ </li>`)
34
+ .join('');
35
+ sendHtml(res, page('Search', `<ul id="results">${rows}</ul>`));
36
+ });
37
+ }
@@ -0,0 +1,52 @@
1
+ import { startFixture, sendHtml, sendJson, sendJs } from './harness.js';
2
+ import { search, DATASET } from './data.js';
3
+ const EDITORS = new Map(DATASET.map((r, i) => [r.id, i % 3 === 1 ? `editor-${i % 5}` : null]));
4
+ function withEditor(r) {
5
+ return { ...r, editor: EDITORS.get(r.id) ?? null };
6
+ }
7
+ const BOOT_SCRIPT = `
8
+ const params = new URLSearchParams(location.search)
9
+ const q = params.get('q') || ''
10
+ const page = params.get('page') || '1'
11
+ async function boot() {
12
+ const res = await fetch('/api/search?q=' + encodeURIComponent(q) + '&page=' + encodeURIComponent(page))
13
+ const data = await res.json()
14
+ const ul = document.createElement('ul')
15
+ ul.id = 'results'
16
+ for (const r of data.results) {
17
+ const li = document.createElement('li')
18
+ li.className = 'result'
19
+ li.dataset.id = r.id
20
+ const a = document.createElement('a')
21
+ a.href = '/item/' + r.id
22
+ a.textContent = r.title + ' by ' + (r.editor ?? r.author)
23
+ li.appendChild(a)
24
+ ul.appendChild(li)
25
+ }
26
+ document.getElementById('app').appendChild(ul)
27
+ }
28
+ boot()
29
+ `;
30
+ export function startCoalesceFixture() {
31
+ return startFixture('siteCoalesce', (req, res) => {
32
+ const url = new URL(req.url ?? '/', 'http://localhost');
33
+ switch (true) {
34
+ case url.pathname === '/search':
35
+ sendHtml(res, `<!doctype html><html><head><title>Search</title>
36
+ <script src="/app.js" defer></script></head>
37
+ <body><header>Site Coalesce</header><div id="app"></div></body></html>`);
38
+ return;
39
+ case url.pathname === '/app.js':
40
+ sendJs(res, BOOT_SCRIPT);
41
+ return;
42
+ case url.pathname === '/api/search': {
43
+ const q = url.searchParams.get('q') ?? '';
44
+ const page = Number(url.searchParams.get('page') ?? '1');
45
+ sendJson(res, { query: q, page, total: DATASET.length, results: search(q, page).map(withEditor) });
46
+ return;
47
+ }
48
+ default:
49
+ sendJson(res, { error: 'not found' }, 404);
50
+ }
51
+ });
52
+ }