webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,29 @@
1
+ import { extractBySelector } from '../executor/extract.js';
2
+ import { diffGolden } from '../benchmark/golden.js';
3
+ /**
4
+ * Checks a candidate recipe's output against what the browser actually
5
+ * extracted from the same visit. This is the golden diff moved to compile
6
+ * time: without it a recipe can point at a plausible-looking endpoint that
7
+ * carries different data — a crate's owners rather than the crate, an
8
+ * article's own URL rather than the discussion link — and nothing notices
9
+ * until the numbers are already wrong.
10
+ */
11
+ /** What the browser extracted from this visit, using the plan's own selectors. */
12
+ export function browserItemsOf(trace, plan) {
13
+ return extractBySelector(trace.finalHtml, plan.itemSelector, plan.fields);
14
+ }
15
+ export function verifyAgainstBrowser(trace, plan, recipeItems) {
16
+ const browserItems = browserItemsOf(trace, plan);
17
+ if (browserItems.length === 0) {
18
+ return { equivalent: false, reasons: ['browser extracted nothing to compare against'] };
19
+ }
20
+ // Two extractions that both found nothing agree vacuously. If the browser
21
+ // side carries no values, the plan's selectors are wrong and there is no
22
+ // behaviour to reproduce.
23
+ const populated = browserItems.some((item) => Object.values(item).some((v) => v !== null && v !== undefined && v !== ''));
24
+ if (!populated) {
25
+ return { equivalent: false, reasons: ['browser extracted items with no field values; check the plan selectors'] };
26
+ }
27
+ const diff = diffGolden({ taskId: 'compile-check', capturedAt: new Date().toISOString(), items: browserItems }, recipeItems);
28
+ return { equivalent: diff.match, reasons: diff.reasons };
29
+ }
@@ -0,0 +1,179 @@
1
+ import * as cheerio from 'cheerio';
2
+ import { resolvePath } from '../recipes/paths.js';
3
+ import { renderRowTemplate } from '../compiler/derive.js';
4
+ /** Strings are trimmed: browser extraction trims, and the two must agree. */
5
+ function coerce(value) {
6
+ if (value === null || value === undefined)
7
+ return null;
8
+ if (typeof value === 'string')
9
+ return value.trim();
10
+ if (typeof value === 'number')
11
+ return value;
12
+ return String(value).trim();
13
+ }
14
+ /** A path that resolves to one object yields a single item; an array yields many. */
15
+ export function extractJsonItems(recipe, payload) {
16
+ if (recipe.output.type !== 'json')
17
+ throw new Error('extractJsonItems requires a json recipe');
18
+ const located = resolvePath(payload, recipe.output.items.path);
19
+ if (located === undefined || located === null)
20
+ return [];
21
+ const rows = Array.isArray(located) ? located : [located];
22
+ const fields = Object.entries(recipe.output.items.fields);
23
+ // A spec starting with `$` reads a value; anything else composes one from the
24
+ // row, reproducing a derivation the page performs on top of its API.
25
+ return rows.map((row) => Object.fromEntries(fields.map(([name, spec]) => [
26
+ name,
27
+ spec.startsWith('$')
28
+ ? coerce(resolvePath(row, spec))
29
+ : renderRowTemplate(spec, row).trim(),
30
+ ])));
31
+ }
32
+ export function extractHtmlItems(recipe, html) {
33
+ if (recipe.output.type !== 'html')
34
+ throw new Error('extractHtmlItems requires an html recipe');
35
+ return extractBySelector(html, recipe.output.items.selector, recipe.output.items.fields);
36
+ }
37
+ /**
38
+ * The structural signature of an HTML response: which of the recipe's declared
39
+ * fields actually resolved. Fingerprinting the raw HTML would change on every
40
+ * content edit; fingerprinting the recipe's own selectors would never change at
41
+ * all. What matters is whether the selectors still find anything.
42
+ */
43
+ export function htmlSignature(recipe, items) {
44
+ if (recipe.output.type !== 'html')
45
+ throw new Error('htmlSignature requires an html recipe');
46
+ const first = items[0];
47
+ const fields = first === undefined
48
+ ? []
49
+ : Object.keys(first).filter((name) => first[name] !== null && first[name] !== '').sort();
50
+ return { selector: recipe.output.items.selector, fields };
51
+ }
52
+ /**
53
+ * The structural signature of a JSON response, limited to what the recipe
54
+ * actually reads. Hashing every path in the payload treats content-driven
55
+ * variation as schema drift — Algolia's per-hit _highlightResult carries
56
+ * different keys for different queries — so a recipe would be judged broken
57
+ * every time the search term changed.
58
+ */
59
+ export function jsonSignature(recipe, payload) {
60
+ const located = resolvePath(payload, recipe.output.items.path);
61
+ const rows = Array.isArray(located) ? located : located === undefined || located === null ? [] : [located];
62
+ const sample = rows[0];
63
+ const fields = sample === undefined
64
+ ? []
65
+ : Object.entries(recipe.output.items.fields)
66
+ .filter(([, spec]) => (spec.startsWith('$') ? resolvePath(sample, spec) !== undefined : true))
67
+ .map(([name]) => name)
68
+ .sort();
69
+ return { itemsPath: located === undefined || located === null ? '(missing)' : recipe.output.items.path, fields };
70
+ }
71
+ /** Elements HTML parsing only accepts inside a particular parent. */
72
+ const FRAGMENT_WRAPPERS = [
73
+ [/^<tr[\s>]/i, (h) => `<table><tbody>${h}</tbody></table>`],
74
+ [/^<(td|th)[\s>]/i, (h) => `<table><tbody><tr>${h}</tr></tbody></table>`],
75
+ [/^<(li)[\s>]/i, (h) => `<ul>${h}</ul>`],
76
+ [/^<(option)[\s>]/i, (h) => `<select>${h}</select>`],
77
+ [/^<(dt|dd)[\s>]/i, (h) => `<dl>${h}</dl>`],
78
+ ];
79
+ /**
80
+ * Parses HTML that may be a fragment rather than a document.
81
+ *
82
+ * A response can be a bare run of `<tr>` elements, which an XHR splices into an
83
+ * existing table — remoteok.com answers a search that way. Parsing that as a
84
+ * document silently discards every row, because a table row outside a table is
85
+ * invalid HTML, and the result is an empty extraction with no error to explain
86
+ * it. Supplying the missing parent keeps the rows.
87
+ */
88
+ export function parseHtmlFragment(html) {
89
+ const head = html.replace(/^\s*(?:<!--[\s\S]*?-->\s*)*/, '');
90
+ const wrapper = FRAGMENT_WRAPPERS.find(([pattern]) => pattern.test(head));
91
+ return cheerio.load(wrapper ? wrapper[1](head) : html);
92
+ }
93
+ /** One interpretation of a field spec, shared by cheerio and the browser. */
94
+ export function planFields(fields) {
95
+ return Object.entries(fields).map(([name, spec]) => {
96
+ if (spec === '')
97
+ return { name, mode: 'own-text' };
98
+ if (spec.startsWith('@'))
99
+ return { name, mode: 'own-attr', attribute: spec.slice(1) };
100
+ const suffix = attributeSuffix(spec);
101
+ return suffix
102
+ ? { name, mode: 'find-attr', selector: suffix.selector, attribute: suffix.attribute }
103
+ : { name, mode: 'find-text', selector: spec };
104
+ });
105
+ }
106
+ /**
107
+ * Splits `selector@attr`, tolerating an `@` inside the selector itself — jsr.io
108
+ * links start with `/@`, so `a[href^="/@"]@href` must split on the trailing
109
+ * suffix and not on the first one it finds. The suffix only counts outside
110
+ * brackets and quotes.
111
+ */
112
+ function attributeSuffix(spec) {
113
+ const match = /@([a-zA-Z][\w-]*)$/.exec(spec);
114
+ if (!match)
115
+ return null;
116
+ const selector = spec.slice(0, match.index);
117
+ let depth = 0;
118
+ let quote = null;
119
+ for (const ch of selector) {
120
+ if (quote) {
121
+ if (ch === quote)
122
+ quote = null;
123
+ continue;
124
+ }
125
+ if (ch === '"' || ch === "'") {
126
+ quote = ch;
127
+ continue;
128
+ }
129
+ if (ch === '[')
130
+ depth += 1;
131
+ else if (ch === ']')
132
+ depth -= 1;
133
+ }
134
+ // An unbalanced selector means the @ we matched belongs inside it.
135
+ if (depth !== 0 || quote !== null || selector === '')
136
+ return null;
137
+ return { selector, attribute: match[1] };
138
+ }
139
+ /**
140
+ * Extracts items from HTML with a bare selector and field map, independent of a
141
+ * recipe. Used to reconstruct what the browser saw so a candidate recipe can be
142
+ * checked against it.
143
+ */
144
+ export function extractBySelector(html, selector, fields) {
145
+ const $ = parseHtmlFragment(html);
146
+ const plans = planFields(fields);
147
+ return $(selector).toArray().map((element) => {
148
+ const item = $(element);
149
+ const row = {};
150
+ // Field candidates ask for `a`, `a@href` and `a@id` of the same item; one find serves all three.
151
+ const found = new Map();
152
+ const find = (sel) => {
153
+ let f = found.get(sel);
154
+ if (f === undefined) {
155
+ f = item.find(sel);
156
+ found.set(sel, f);
157
+ }
158
+ return f;
159
+ };
160
+ for (const f of plans) {
161
+ switch (f.mode) {
162
+ case 'own-text':
163
+ row[f.name] = item.text().trim();
164
+ break;
165
+ case 'own-attr':
166
+ row[f.name] = item.attr(f.attribute) ?? null;
167
+ break;
168
+ case 'find-attr':
169
+ row[f.name] = find(f.selector).attr(f.attribute) ?? null;
170
+ break;
171
+ default: {
172
+ const matches = find(f.selector);
173
+ row[f.name] = matches.length === 0 ? null : matches.first().text().trim();
174
+ }
175
+ }
176
+ }
177
+ return row;
178
+ });
179
+ }
@@ -0,0 +1,55 @@
1
+ const ORIGIN = /^https?:\/\/[^/]+/;
2
+ /** The `https?://host` every item's url starts with, or undefined if they do not share one. */
3
+ function sharedOrigin(items) {
4
+ if (items.length === 0)
5
+ return undefined;
6
+ let origin;
7
+ for (const item of items) {
8
+ const url = item.url;
9
+ if (typeof url !== 'string')
10
+ return undefined;
11
+ const match = ORIGIN.exec(url);
12
+ if (match === null)
13
+ return undefined;
14
+ if (origin === undefined)
15
+ origin = match[0];
16
+ else if (origin !== match[0])
17
+ return undefined;
18
+ }
19
+ return origin;
20
+ }
21
+ // A tab, newline or carriage return inside a value becomes a space, so a title
22
+ // cannot break the table. That is the whole escaping story: no quoting.
23
+ function cell(value) {
24
+ if (value === null || value === undefined)
25
+ return '';
26
+ return String(value).replace(/[\t\n\r]/g, ' ');
27
+ }
28
+ /**
29
+ * Serialises items for an agent to read. `tsv` is a header line plus one line
30
+ * per item, preceded by a `base` line carrying the origin every url shares.
31
+ */
32
+ export function formatItems(items, format) {
33
+ if (format === 'json')
34
+ return JSON.stringify(items);
35
+ const columns = [];
36
+ for (const item of items) {
37
+ for (const key of Object.keys(item))
38
+ if (!columns.includes(key))
39
+ columns.push(key);
40
+ }
41
+ const origin = sharedOrigin(items);
42
+ const lines = origin === undefined ? [] : [`base\t${origin}`];
43
+ lines.push(columns.join('\t'));
44
+ for (const item of items) {
45
+ lines.push(columns
46
+ .map((key) => {
47
+ const value = item[key];
48
+ return key === 'url' && origin !== undefined && typeof value === 'string'
49
+ ? cell(value.slice(origin.length))
50
+ : cell(value);
51
+ })
52
+ .join('\t'));
53
+ }
54
+ return lines.join('\n');
55
+ }
@@ -0,0 +1,147 @@
1
+ import { measureResult } from '../measurement.js';
2
+ import { validate } from '../validator/index.js';
3
+ import { addMeta } from '../types.js';
4
+ /** How long a site stays off the recipe path after it refused one. */
5
+ export const BLOCK_COOLDOWN_MS = 10 * 60_000;
6
+ /** Statuses a site uses to say no: auth wall, forbidden, rate limit, shed load. */
7
+ const BLOCK_STATUSES = new Set([401, 403, 429, 503]);
8
+ /** Replaced in place once the healer reports why the compiler refused. */
9
+ const NO_RECIPE = 'no recipe registered';
10
+ /** Only a json signature reports whether the recipe's items path still resolved. */
11
+ function keptRecordedShape(recipe, payload) {
12
+ // An html recipe has no items path to test, so its challenge pages are caught
13
+ // only because this returns false; narrowing this line away kills rule (b) for
14
+ // every html site, with nothing but the siteRefusing test to notice.
15
+ if (recipe.output.type !== 'json')
16
+ return false;
17
+ return payload?.itemsPath === recipe.output.items.path;
18
+ }
19
+ export class Executor {
20
+ opts;
21
+ /** Site -> the time until which it is treated as refusing us. */
22
+ blockedUntil = new Map();
23
+ constructor(opts) {
24
+ this.opts = opts;
25
+ }
26
+ strategy(name) {
27
+ const found = this.opts.strategies.find((s) => s.name === name);
28
+ if (!found)
29
+ throw new Error(`no strategy registered for "${name}"`);
30
+ return found;
31
+ }
32
+ /**
33
+ * Descends the browser levels rather than jumping to the bottom. A warm pool
34
+ * still needs a browser but does not pay to start one, and hardcoding the
35
+ * cold browser here made that distinction unobservable: the level split
36
+ * reported L2 at zero because nothing could ever reach it.
37
+ */
38
+ async fallback(recipe, task, reasons) {
39
+ const ladder = this.opts.strategies.filter((s) => s.name === 'warm-browser' || s.name === 'browser');
40
+ ladder.sort((a, b) => (a.name === 'warm-browser' ? -1 : 1) - (b.name === 'warm-browser' ? -1 : 1));
41
+ let lastError = new Error('no browser strategy registered');
42
+ for (const strategy of ladder) {
43
+ try {
44
+ const result = await strategy.execute(recipe ?? {}, task);
45
+ return { ...result, recipeUsed: false, fellBack: true, reasons };
46
+ }
47
+ catch (err) {
48
+ lastError = err;
49
+ reasons.push(err instanceof Error ? err.message : String(err));
50
+ }
51
+ }
52
+ throw lastError;
53
+ }
54
+ async run(task) {
55
+ return measureResult('browser', () => this.runTask(task));
56
+ }
57
+ async runTask(task) {
58
+ const now = this.opts.now ?? Date.now;
59
+ const recipe = await this.opts.registry.load(task.site, task.intent);
60
+ const reasons = [];
61
+ const cooldown = this.blockedUntil.get(task.site);
62
+ if (cooldown !== undefined && cooldown > now()) {
63
+ reasons.push(`blocked: cooling down for ${task.site}`);
64
+ const cooling = await this.fallback(recipe, task, reasons);
65
+ return { ...cooling, blocked: true, reasons: [...reasons] };
66
+ }
67
+ let attempt = null;
68
+ /** What was spent before the answer arrived, kept so the total can include it. */
69
+ const spent = [];
70
+ if (recipe && recipe.strategy.type !== 'browser') {
71
+ try {
72
+ const result = await this.strategy(recipe.strategy.type).execute(recipe, task);
73
+ const outcome = validate(recipe, {
74
+ status: result.status ?? recipe.validation.status,
75
+ payload: result.payload,
76
+ items: result.items,
77
+ });
78
+ if (outcome.valid) {
79
+ return { ...result, recipeUsed: true, fellBack: false, blocked: false, reasons: [] };
80
+ }
81
+ reasons.push(...outcome.reasons);
82
+ spent.push(result.meta);
83
+ // The raw status, never the expected one standing in for it: reading a
84
+ // missing status as the expected one would make every browser result a
85
+ // 200 that carried nothing, which is half of rule (b).
86
+ if (result.status !== undefined) {
87
+ attempt = {
88
+ status: result.status,
89
+ items: result.items,
90
+ recordedShape: keptRecordedShape(recipe, result.payload),
91
+ };
92
+ }
93
+ }
94
+ catch (err) {
95
+ reasons.push(err instanceof Error ? err.message : String(err));
96
+ }
97
+ }
98
+ else if (!recipe) {
99
+ reasons.push(NO_RECIPE);
100
+ }
101
+ // (a) is settled before the browser runs. A site that answered 403 has
102
+ // earned the cooldown whatever the browser then does, and deciding it after
103
+ // would lose the entry whenever the fallback throws.
104
+ let blocked = false;
105
+ if (attempt && BLOCK_STATUSES.has(attempt.status)) {
106
+ reasons.push(`blocked: status ${attempt.status}`);
107
+ this.blockedUntil.set(task.site, now() + BLOCK_COOLDOWN_MS);
108
+ blocked = true;
109
+ }
110
+ const fellBack = await this.fallback(recipe, task, reasons);
111
+ // (b) cannot be settled any earlier: a 200 that carried nothing is only a
112
+ // challenge page if the browser beside it could not read the page either.
113
+ // Drift the browser can still read is a rotted recipe, and heals.
114
+ // An empty result and a challenge page both carry no items; only the empty
115
+ // result still has the response shape the recipe recorded.
116
+ if (!blocked && attempt?.status === 200 && attempt.items.length === 0
117
+ && !attempt.recordedShape && fellBack.items.length === 0) {
118
+ reasons.push('blocked: 200 with no items, browser also empty');
119
+ this.blockedUntil.set(task.site, now() + BLOCK_COOLDOWN_MS);
120
+ blocked = true;
121
+ }
122
+ // (c) costs this one task a re-record, and has to: comparing the recipe the
123
+ // browser yields now against the stored one is the measurement, and there
124
+ // is no way to learn it changed nothing without performing it once.
125
+ if (!blocked) {
126
+ const healResult = await this.opts.onFallback?.({ task, recipe, reasons });
127
+ if (healResult?.meta)
128
+ spent.push(healResult.meta);
129
+ // On the run that learned it, not a later one: a benchmark task runs
130
+ // once, so a reason held back for next time is a reason nothing reports.
131
+ if (healResult?.refused !== undefined) {
132
+ const bare = reasons.indexOf(NO_RECIPE);
133
+ const line = `no recipe: ${healResult.refused}`;
134
+ if (bare === -1)
135
+ reasons.push(line);
136
+ else
137
+ reasons[bare] = line;
138
+ }
139
+ if (healResult && healResult.unchanged) {
140
+ reasons.push('blocked: re-record produced the same recipe — the site serves the engine a different page');
141
+ this.blockedUntil.set(task.site, now() + BLOCK_COOLDOWN_MS);
142
+ blocked = true;
143
+ }
144
+ }
145
+ return { ...fellBack, meta: addMeta(fellBack.meta, ...spent), blocked, reasons: [...reasons] };
146
+ }
147
+ }
@@ -0,0 +1,60 @@
1
+ import { measureResult } from '../../measurement.js';
2
+ import { openSession } from '../../browser/session.js';
3
+ import { navigateAndSettle } from '../../browser/navigate.js';
4
+ import { planFields } from '../extract.js';
5
+ import { countTokens } from '../tokens.js';
6
+ import { emptyMeta } from '../../types.js';
7
+ export const BROWSER_PLANS = {};
8
+ export class BrowserStrategy {
9
+ sites;
10
+ plans;
11
+ countAgentTokens;
12
+ name = 'browser';
13
+ constructor(sites, plans = BROWSER_PLANS,
14
+ /** The token count costs a page read of its own; a timing benchmark can decline to pay it. */
15
+ countAgentTokens = true) {
16
+ this.sites = sites;
17
+ this.plans = plans;
18
+ this.countAgentTokens = countAgentTokens;
19
+ }
20
+ async execute(recipe, task) {
21
+ return measureResult(this.name, () => this.executeAttempt(recipe, task));
22
+ }
23
+ async executeAttempt(_recipe, task) {
24
+ const plan = this.plans[task.site]?.[task.intent];
25
+ if (!plan)
26
+ throw new Error(`no browser plan for ${task.site}/${task.intent}`);
27
+ const meta = emptyMeta(this.name);
28
+ const started = performance.now();
29
+ const session = await openSession();
30
+ meta.browserLaunches = 1;
31
+ try {
32
+ await navigateAndSettle(session.page, plan.url(this.sites.origin(task.site), task), plan.itemSelector);
33
+ // The field specs are interpreted once, in node, so that the browser and
34
+ // cheerio cannot drift apart on what a spec means.
35
+ const items = (await session.page.$$eval(plan.itemSelector, (elements, plans) => elements.map((el) => Object.fromEntries(plans.map((f) => {
36
+ switch (f.mode) {
37
+ case 'own-text': return [f.name, el.textContent?.trim() ?? null];
38
+ case 'own-attr': return [f.name, el.getAttribute(f.attribute)];
39
+ case 'find-attr': return [f.name, el.querySelector(f.selector)?.getAttribute(f.attribute) ?? null];
40
+ default: return [f.name, el.querySelector(f.selector)?.textContent?.trim() ?? null];
41
+ }
42
+ }))), planFields(plan.fields)));
43
+ meta.pageNavigations = session.cost.pageNavigations;
44
+ meta.networkRequests = session.cost.networkRequests;
45
+ meta.bytesDownloaded = session.cost.bytesDownloaded;
46
+ meta.latencyMs = Math.round(performance.now() - started);
47
+ // After the latency line: taking the snapshot and tokenising it both take
48
+ // real time, and charging that to the baseline would flatter the engine.
49
+ // The read is allowed to fail — a late client-side navigation destroys
50
+ // the execution context — because instrumentation must never turn a
51
+ // graded run into a thrown one.
52
+ const snapshot = this.countAgentTokens ? await session.page.locator('body').ariaSnapshot().catch(() => '') : '';
53
+ meta.llmTokens = countTokens(snapshot);
54
+ return { items, meta };
55
+ }
56
+ finally {
57
+ await session.close();
58
+ }
59
+ }
60
+ }
@@ -0,0 +1,42 @@
1
+ import { measureResult } from '../../measurement.js';
2
+ import { render } from '../../recipes/template.js';
3
+ import { buildUrl } from './http-json.js';
4
+ import { extractHtmlItems, htmlSignature } from '../extract.js';
5
+ import { countTokens } from '../tokens.js';
6
+ import { formatItems } from '../format.js';
7
+ import { emptyMeta } from '../../types.js';
8
+ export class HttpHtmlStrategy {
9
+ net;
10
+ sites;
11
+ name = 'http-html';
12
+ constructor(net, sites) {
13
+ this.net = net;
14
+ this.sites = sites;
15
+ }
16
+ async execute(recipe, task) {
17
+ return measureResult(this.name, () => this.executeAttempt(recipe, task));
18
+ }
19
+ async executeAttempt(recipe, task) {
20
+ const meta = emptyMeta(this.name);
21
+ const started = performance.now();
22
+ const url = buildUrl(this.sites.origin(task.site), recipe, task);
23
+ const res = await this.net.fetch(url, {
24
+ method: recipe.request.method,
25
+ headers: recipe.request.headers,
26
+ body: recipe.request.body ? render(recipe.request.body, task.input) : undefined,
27
+ });
28
+ meta.networkRequests = 1;
29
+ meta.bytesDownloaded = res.bytesDownloaded;
30
+ meta.politenessWaitMs = res.waitedMs;
31
+ meta.latencyMs = Math.max(0, Math.round(performance.now() - started) - res.waitedMs);
32
+ // Reported, not thrown. A status that becomes an error message is invisible
33
+ // to the validator that has to judge it and to the executor that has to tell
34
+ // a site refusing us from a recipe that rotted.
35
+ if (res.status !== recipe.validation.status) {
36
+ return { items: [], meta, status: res.status, payload: undefined };
37
+ }
38
+ const items = extractHtmlItems(recipe, res.body);
39
+ meta.llmTokens = countTokens(formatItems(items, 'tsv'));
40
+ return { items, meta, payload: htmlSignature(recipe, items), status: res.status };
41
+ }
42
+ }
@@ -0,0 +1,71 @@
1
+ import { measureResult } from '../../measurement.js';
2
+ import { render, renderPath, renderQuery } from '../../recipes/template.js';
3
+ import { extractJsonItems, jsonSignature } from '../extract.js';
4
+ import { countTokens } from '../tokens.js';
5
+ import { formatItems } from '../format.js';
6
+ import { emptyMeta } from '../../types.js';
7
+ export function buildUrl(siteOrigin, recipe, task) {
8
+ const origin = new URL(recipe.request.origin ?? siteOrigin).origin;
9
+ const url = new URL(renderPath(recipe.request.path, task.input), origin);
10
+ // An input value must not be able to redirect a request the recipe did not
11
+ // describe: a rendered path beginning `//`, such as a `detail` input of
12
+ // `/evil.example/x` against a path of `/{{id}}`, is protocol-relative and
13
+ // `new URL` resolves it against that host instead of `origin`.
14
+ if (url.origin !== origin)
15
+ throw new Error(`recipe would request ${url.origin}, outside ${origin}`);
16
+ for (const [k, v] of Object.entries(renderQuery(recipe.request.query ?? {}, task.input))) {
17
+ url.searchParams.set(k, v);
18
+ }
19
+ return url.toString();
20
+ }
21
+ export class HttpJsonStrategy {
22
+ net;
23
+ sites;
24
+ name = 'http-json';
25
+ constructor(net, sites) {
26
+ this.net = net;
27
+ this.sites = sites;
28
+ }
29
+ async execute(recipe, task) {
30
+ return measureResult(this.name, () => this.executeAttempt(recipe, task));
31
+ }
32
+ async executeAttempt(recipe, task) {
33
+ const meta = emptyMeta(this.name);
34
+ const started = performance.now();
35
+ const url = buildUrl(this.sites.origin(task.site), recipe, task);
36
+ const res = await this.net.fetch(url, {
37
+ method: recipe.request.method,
38
+ headers: recipe.request.headers,
39
+ body: recipe.request.body ? render(recipe.request.body, task.input) : undefined,
40
+ });
41
+ meta.networkRequests = 1;
42
+ meta.bytesDownloaded = res.bytesDownloaded;
43
+ meta.politenessWaitMs = res.waitedMs;
44
+ meta.latencyMs = Math.max(0, Math.round(performance.now() - started) - res.waitedMs);
45
+ // Reported, not thrown. A status that becomes an error message is invisible
46
+ // to the validator that has to judge it and to the executor that has to tell
47
+ // a site refusing us from a recipe that rotted.
48
+ if (res.status !== recipe.validation.status) {
49
+ return { items: [], meta, status: res.status, payload: undefined };
50
+ }
51
+ let payload;
52
+ try {
53
+ payload = JSON.parse(res.body);
54
+ }
55
+ catch {
56
+ // A challenge page answers 200 with HTML where the API used to be; that is
57
+ // a site response like any other and is reported the same way.
58
+ return { items: [], meta, status: res.status, payload: undefined };
59
+ }
60
+ if (recipe.output.type !== 'json')
61
+ throw new Error('http-json requires a json recipe');
62
+ const items = extractJsonItems(recipe, payload);
63
+ meta.llmTokens = countTokens(formatItems(items, 'tsv'));
64
+ return {
65
+ items,
66
+ meta,
67
+ payload: jsonSignature({ output: recipe.output }, payload),
68
+ status: res.status,
69
+ };
70
+ }
71
+ }
@@ -0,0 +1,57 @@
1
+ import { measureResult } from '../../measurement.js';
2
+ import { BrowserPool } from '../../browser/pool.js';
3
+ import { navigateAndSettle } from '../../browser/navigate.js';
4
+ import { planFields } from '../extract.js';
5
+ import { countTokens } from '../tokens.js';
6
+ import { BROWSER_PLANS } from './browser.js';
7
+ import { emptyMeta } from '../../types.js';
8
+ /** A browser run that does not pay to start a browser. */
9
+ export class WarmBrowserStrategy {
10
+ sites;
11
+ plans;
12
+ name = 'warm-browser';
13
+ pool = new BrowserPool();
14
+ constructor(sites, plans = BROWSER_PLANS) {
15
+ this.sites = sites;
16
+ this.plans = plans;
17
+ }
18
+ async execute(recipe, task) {
19
+ return measureResult(this.name, () => this.executeAttempt(recipe, task));
20
+ }
21
+ async executeAttempt(_recipe, task) {
22
+ const plan = this.plans[task.site]?.[task.intent];
23
+ if (!plan)
24
+ throw new Error(`no browser plan for ${task.site}/${task.intent}`);
25
+ const meta = emptyMeta(this.name);
26
+ const started = performance.now();
27
+ const warm = await this.pool.acquire();
28
+ try {
29
+ await navigateAndSettle(warm.page, plan.url(this.sites.origin(task.site), task), plan.itemSelector);
30
+ const items = (await warm.page.$$eval(plan.itemSelector, (elements, plans) => elements.map((el) => Object.fromEntries(plans.map((f) => {
31
+ switch (f.mode) {
32
+ case 'own-text': return [f.name, el.textContent?.trim() ?? null];
33
+ case 'own-attr': return [f.name, el.getAttribute(f.attribute)];
34
+ case 'find-attr': return [f.name, el.querySelector(f.selector)?.getAttribute(f.attribute) ?? null];
35
+ default: return [f.name, el.querySelector(f.selector)?.textContent?.trim() ?? null];
36
+ }
37
+ }))), planFields(plan.fields)));
38
+ // Zero only when the pool already had a browser. The first call starts
39
+ // Chromium like any cold run does, and reporting none said otherwise.
40
+ meta.browserLaunches = warm.launched ? 1 : 0;
41
+ meta.pageNavigations = warm.cost.pageNavigations;
42
+ meta.networkRequests = warm.cost.networkRequests;
43
+ meta.bytesDownloaded = warm.cost.bytesDownloaded;
44
+ meta.latencyMs = Math.round(performance.now() - started);
45
+ // After the latency line, and allowed to fail, for the reasons in browser.ts.
46
+ const snapshot = await warm.page.locator('body').ariaSnapshot().catch(() => '');
47
+ meta.llmTokens = countTokens(snapshot);
48
+ return { items, meta };
49
+ }
50
+ finally {
51
+ await warm.release();
52
+ }
53
+ }
54
+ async close() {
55
+ await this.pool.close();
56
+ }
57
+ }
@@ -0,0 +1,11 @@
1
+ import { Tiktoken } from 'js-tiktoken';
2
+ import cl100k_base from 'js-tiktoken/ranks/cl100k_base';
3
+ // The rank table is a static import and loads with this module; what is
4
+ // deferred is turning it into the encoder's lookup maps, which is the
5
+ // expensive half and is done once, on first use.
6
+ let encoder;
7
+ /** Tokens an agent would spend reading `text`, under cl100k_base. */
8
+ export function countTokens(text) {
9
+ encoder ??= new Tiktoken(cl100k_base);
10
+ return encoder.encode(text).length;
11
+ }