webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
package/dist/src/read.js
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
import { load } from 'cheerio';
|
|
2
|
+
import { readFile } from 'node:fs/promises';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { UserError, atomicWrite } from './local.js';
|
|
5
|
+
// Below this much visible text the server HTML is treated as a shell for a script to fill.
|
|
6
|
+
const MIN_TEXT = 200;
|
|
7
|
+
const BLOCK_STATUSES = new Set([401, 403, 429, 503]);
|
|
8
|
+
/** Host plus path, with id- and slug-like segments collapsed, so pages of one structure share a key. */
|
|
9
|
+
export function urlShape(url) {
|
|
10
|
+
const u = new URL(url);
|
|
11
|
+
const segments = u.pathname.split('/').filter(Boolean)
|
|
12
|
+
.map(s => /\d/.test(s) || s.split('-').length >= 3 || s.length >= 16 ? '*' : s);
|
|
13
|
+
return `${u.host}/${segments.join('/')}`;
|
|
14
|
+
}
|
|
15
|
+
const HIDDEN = 'script, style, noscript, template, svg, iframe';
|
|
16
|
+
// Site chrome repeats on every page and can outweigh the content many times over.
|
|
17
|
+
const CHROME = 'nav, header, footer, aside, [role=navigation], [role=banner], [role=contentinfo]';
|
|
18
|
+
const BLOCKS = 'p, div, li, dt, dd, h1, h2, h3, h4, h5, h6, br, tr, section, article, main, ul, ol, table, pre, blockquote, figcaption';
|
|
19
|
+
export function visibleText(html) {
|
|
20
|
+
const $ = load(html);
|
|
21
|
+
const title = $('title').first().text().trim();
|
|
22
|
+
$(HIDDEN).remove();
|
|
23
|
+
$(CHROME).remove();
|
|
24
|
+
$(BLOCKS).after('\n');
|
|
25
|
+
const main = $('main, [role=main], article').first();
|
|
26
|
+
const root = main.length && main.text().trim() ? main : $('body');
|
|
27
|
+
const text = root.text().replace(/[ \t\f\v\r]+/g, ' ').replace(/\s*\n\s*/g, '\n').trim();
|
|
28
|
+
return { title, text };
|
|
29
|
+
}
|
|
30
|
+
export async function readUrl(url, deps) {
|
|
31
|
+
if (!(await deps.isAllowed(url)))
|
|
32
|
+
throw new UserError('ROBOTS_DISALLOWED', `robots.txt disallows ${url}`);
|
|
33
|
+
const shape = urlShape(url);
|
|
34
|
+
let reason = 'remembered: this URL shape needed a browser before';
|
|
35
|
+
if ((await deps.methods.get(shape)) !== 'browser') {
|
|
36
|
+
const res = await deps.fetch(url);
|
|
37
|
+
const type = res.headers['content-type'] ?? '';
|
|
38
|
+
if (res.status >= 200 && res.status < 300 && !/html/i.test(type)) {
|
|
39
|
+
await deps.methods.set(shape, 'http');
|
|
40
|
+
return { url, method: 'http', reason: `server returned ${type || 'a non-HTML body'}`, title: '', text: res.body };
|
|
41
|
+
}
|
|
42
|
+
if (res.status >= 200 && res.status < 300) {
|
|
43
|
+
const page = visibleText(res.body);
|
|
44
|
+
if (page.text.length >= MIN_TEXT) {
|
|
45
|
+
await deps.methods.set(shape, 'http');
|
|
46
|
+
return { url, method: 'http', reason: 'server HTML carries the text; not the rendered page', ...page };
|
|
47
|
+
}
|
|
48
|
+
reason = `server HTML is a shell (${page.text.length} chars of text)`;
|
|
49
|
+
}
|
|
50
|
+
else if (BLOCK_STATUSES.has(res.status)) {
|
|
51
|
+
reason = `HTTP ${res.status}`;
|
|
52
|
+
}
|
|
53
|
+
else {
|
|
54
|
+
throw new UserError('EXECUTION_FAILED', `HTTP ${res.status} for ${url}`);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
const page = visibleText(await deps.render(url));
|
|
58
|
+
if (page.text.length === 0)
|
|
59
|
+
throw new UserError('UNVERIFIED_RESULT', `the rendered page has no text: an empty page, an access block, or content that needs interaction (${reason})`);
|
|
60
|
+
await deps.methods.set(shape, 'browser');
|
|
61
|
+
return { url, method: 'browser', reason, ...page };
|
|
62
|
+
}
|
|
63
|
+
/** Remembered methods per URL shape, one small JSON file. */
|
|
64
|
+
export function methodStore(root) {
|
|
65
|
+
const path = join(root, 'read-methods.json');
|
|
66
|
+
const load = async () => {
|
|
67
|
+
try {
|
|
68
|
+
return JSON.parse(await readFile(path, 'utf8'));
|
|
69
|
+
}
|
|
70
|
+
catch {
|
|
71
|
+
return {};
|
|
72
|
+
}
|
|
73
|
+
};
|
|
74
|
+
return {
|
|
75
|
+
get: async (shape) => (await load())[shape],
|
|
76
|
+
set: async (shape, method) => {
|
|
77
|
+
const all = await load();
|
|
78
|
+
if (all[shape] === method)
|
|
79
|
+
return;
|
|
80
|
+
await atomicWrite(path, JSON.stringify({ ...all, [shape]: method }, null, 1));
|
|
81
|
+
},
|
|
82
|
+
};
|
|
83
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
const SEPARATOR = '::';
|
|
3
|
+
/**
|
|
4
|
+
* Structural paths of a payload. Arrays collapse to `key[]` and every element
|
|
5
|
+
* is merged, so a schema change is visible but a data change is not.
|
|
6
|
+
*/
|
|
7
|
+
export function collectPaths(value, prefix = '') {
|
|
8
|
+
const out = new Set();
|
|
9
|
+
const walk = (node, path) => {
|
|
10
|
+
if (Array.isArray(node)) {
|
|
11
|
+
if (path !== '')
|
|
12
|
+
out.add(path);
|
|
13
|
+
const arrayPath = `${path}[]`;
|
|
14
|
+
out.add(arrayPath);
|
|
15
|
+
for (const element of node)
|
|
16
|
+
walk(element, arrayPath);
|
|
17
|
+
return;
|
|
18
|
+
}
|
|
19
|
+
if (node !== null && typeof node === 'object') {
|
|
20
|
+
if (path !== '')
|
|
21
|
+
out.add(path);
|
|
22
|
+
for (const [key, child] of Object.entries(node)) {
|
|
23
|
+
walk(child, path === '' ? key : `${path}.${key}`);
|
|
24
|
+
}
|
|
25
|
+
return;
|
|
26
|
+
}
|
|
27
|
+
if (path !== '')
|
|
28
|
+
out.add(path);
|
|
29
|
+
};
|
|
30
|
+
walk(value, prefix);
|
|
31
|
+
return [...out];
|
|
32
|
+
}
|
|
33
|
+
export function computeFingerprint(endpoint, sample) {
|
|
34
|
+
const paths = collectPaths(sample).sort();
|
|
35
|
+
return createHash('sha256')
|
|
36
|
+
.update(endpoint)
|
|
37
|
+
.update(SEPARATOR)
|
|
38
|
+
.update(paths.join('\n'))
|
|
39
|
+
.digest('hex')
|
|
40
|
+
.slice(0, 16);
|
|
41
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/** Minimal JSONPath subset: `$`, `$.key`, `$.a.b`. Enough for recipe output specs. */
|
|
2
|
+
export function resolvePath(root, path) {
|
|
3
|
+
if (path === '$')
|
|
4
|
+
return root;
|
|
5
|
+
if (!path.startsWith('$.'))
|
|
6
|
+
throw new Error(`unsupported path: ${path}`);
|
|
7
|
+
let current = root;
|
|
8
|
+
for (const key of path.slice(2).split('.')) {
|
|
9
|
+
if (current === null || typeof current !== 'object')
|
|
10
|
+
return undefined;
|
|
11
|
+
current = current[key];
|
|
12
|
+
}
|
|
13
|
+
return current;
|
|
14
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { atomicWrite, siteName, intentName } from '../local.js';
|
|
2
|
+
import { mkdir, readFile, readdir, rename, rm } from 'node:fs/promises';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { parse, stringify } from 'yaml';
|
|
5
|
+
import { RecipeSchema } from './schema.js';
|
|
6
|
+
export class RecipeRegistry {
|
|
7
|
+
root;
|
|
8
|
+
constructor(root) {
|
|
9
|
+
this.root = root;
|
|
10
|
+
}
|
|
11
|
+
dir(site) { return join(this.root, siteName(site)); }
|
|
12
|
+
file(site, intent) { return join(this.dir(site), `${intentName(intent)}.yaml`); }
|
|
13
|
+
async load(site, intent) {
|
|
14
|
+
let raw;
|
|
15
|
+
try {
|
|
16
|
+
raw = await readFile(this.file(site, intent), 'utf8');
|
|
17
|
+
}
|
|
18
|
+
catch (err) {
|
|
19
|
+
if (err.code === 'ENOENT')
|
|
20
|
+
return null;
|
|
21
|
+
throw err;
|
|
22
|
+
}
|
|
23
|
+
return RecipeSchema.parse(parse(raw));
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Saving over an existing recipe archives it as `<intent>.<timestamp>.yaml`,
|
|
27
|
+
* unless the recipe is the one already stored.
|
|
28
|
+
*/
|
|
29
|
+
async save(recipe) {
|
|
30
|
+
const parsed = RecipeSchema.parse(recipe);
|
|
31
|
+
await mkdir(this.dir(parsed.site), { recursive: true });
|
|
32
|
+
const target = this.file(parsed.site, parsed.intent);
|
|
33
|
+
const body = stringify(parsed);
|
|
34
|
+
try {
|
|
35
|
+
const stored = await readFile(target, 'utf8');
|
|
36
|
+
// An identical recipe is not a new version: archiving it buries the
|
|
37
|
+
// versions that do differ under copies of the one still in use, which is
|
|
38
|
+
// what a site serving the engine a different page produced every task.
|
|
39
|
+
if (stored === body)
|
|
40
|
+
return target;
|
|
41
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, '-');
|
|
42
|
+
await rename(target, join(this.dir(parsed.site), `${parsed.intent}.${stamp}.yaml`));
|
|
43
|
+
}
|
|
44
|
+
catch (err) {
|
|
45
|
+
if (err.code !== 'ENOENT')
|
|
46
|
+
throw err;
|
|
47
|
+
}
|
|
48
|
+
await atomicWrite(target, body);
|
|
49
|
+
return target;
|
|
50
|
+
}
|
|
51
|
+
async remove(site, intent) {
|
|
52
|
+
await rm(this.file(site, intent), { force: true });
|
|
53
|
+
}
|
|
54
|
+
failureFile(site, intent) { return join(this.dir(site), `${intentName(intent)}.failure.json`); }
|
|
55
|
+
async learningFailure(site, intent) {
|
|
56
|
+
try {
|
|
57
|
+
const state = JSON.parse(await readFile(this.failureFile(site, intent), 'utf8'));
|
|
58
|
+
return typeof state.reason === 'string' && state.until > Date.now() ? state.reason : null;
|
|
59
|
+
}
|
|
60
|
+
catch (error) {
|
|
61
|
+
if (error.code === 'ENOENT')
|
|
62
|
+
return null;
|
|
63
|
+
throw error;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
async setLearningFailure(site, intent, reason) {
|
|
67
|
+
await atomicWrite(this.failureFile(site, intent), JSON.stringify({ reason, until: Date.now() + 24 * 60 * 60_000 }));
|
|
68
|
+
}
|
|
69
|
+
async clearLearningFailure(site, intent) {
|
|
70
|
+
await rm(this.failureFile(site, intent), { force: true });
|
|
71
|
+
}
|
|
72
|
+
async versions(site, intent) {
|
|
73
|
+
try {
|
|
74
|
+
const entries = await readdir(this.dir(site));
|
|
75
|
+
return entries.filter((f) => f.startsWith(`${intent}.`) && f.endsWith('.yaml') && f !== `${intentName(intent)}.yaml`).sort();
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
return [];
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export const RecipeSchema = z.object({
|
|
3
|
+
site: z.string().min(1),
|
|
4
|
+
intent: z.enum(['search', 'list', 'detail']),
|
|
5
|
+
inputs: z.record(z.object({ type: z.enum(['string', 'number']) })),
|
|
6
|
+
strategy: z.object({ type: z.enum(['http-json', 'http-html', 'browser']) }),
|
|
7
|
+
request: z.object({
|
|
8
|
+
method: z.enum(['GET', 'POST']),
|
|
9
|
+
/** Set when the data request lives on another host than the site, e.g. a search vendor. */
|
|
10
|
+
origin: z.string().url().optional(),
|
|
11
|
+
path: z.string(),
|
|
12
|
+
query: z.record(z.string()).optional(),
|
|
13
|
+
headers: z.record(z.string()).optional(),
|
|
14
|
+
body: z.string().optional(),
|
|
15
|
+
}),
|
|
16
|
+
output: z.discriminatedUnion('type', [
|
|
17
|
+
z.object({
|
|
18
|
+
type: z.literal('json'),
|
|
19
|
+
items: z.object({ path: z.string(), fields: z.record(z.string()) }),
|
|
20
|
+
}),
|
|
21
|
+
z.object({
|
|
22
|
+
type: z.literal('html'),
|
|
23
|
+
/** A field value of `@attr` reads that attribute off the item element; anything else is a CSS selector. */
|
|
24
|
+
items: z.object({ selector: z.string(), fields: z.record(z.string()) }),
|
|
25
|
+
}),
|
|
26
|
+
]),
|
|
27
|
+
validation: z.object({
|
|
28
|
+
status: z.number().int(),
|
|
29
|
+
required: z.array(z.string()),
|
|
30
|
+
minItems: z.number().int().default(1),
|
|
31
|
+
}),
|
|
32
|
+
fingerprint: z.object({
|
|
33
|
+
endpoint: z.string(),
|
|
34
|
+
hash: z.string(),
|
|
35
|
+
responseFields: z.array(z.string()),
|
|
36
|
+
}),
|
|
37
|
+
fallback: z.object({ type: z.literal('browser') }),
|
|
38
|
+
});
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { UserError } from '../local.js';
|
|
2
|
+
const PLACEHOLDER = /\{\{\s*([a-zA-Z0-9_]+)\s*\}\}/g;
|
|
3
|
+
export function render(template, input) {
|
|
4
|
+
return template.replace(PLACEHOLDER, (_match, key) => {
|
|
5
|
+
if (!(key in input))
|
|
6
|
+
throw new Error(`no value supplied for placeholder "${key}"`);
|
|
7
|
+
return String(input[key]);
|
|
8
|
+
});
|
|
9
|
+
}
|
|
10
|
+
export function renderQuery(query, input) {
|
|
11
|
+
return Object.fromEntries(Object.entries(query).map(([k, v]) => [k, render(v, input)]));
|
|
12
|
+
}
|
|
13
|
+
/** Encode user values as URL data, retaining slash-separated detail identifiers. */
|
|
14
|
+
export function renderPath(template, input) {
|
|
15
|
+
return render(template, Object.fromEntries(Object.entries(input).map(([k, v]) => [k, String(v).split('/').map(encodeURIComponent).join('/')])));
|
|
16
|
+
}
|
|
17
|
+
export function renderUrlTemplate(origin, template, input) {
|
|
18
|
+
const accepted = new Set([...template.matchAll(PLACEHOLDER)].map(match => match[1]));
|
|
19
|
+
for (const key of Object.keys(input)) {
|
|
20
|
+
if (!accepted.has(key))
|
|
21
|
+
throw new UserError('INVALID_INPUT', `The saved task does not use input "${key}". Teach a URL containing that input first.`);
|
|
22
|
+
}
|
|
23
|
+
const at = template.indexOf('?');
|
|
24
|
+
const path = at === -1 ? template : template.slice(0, at);
|
|
25
|
+
const url = new URL(renderPath(path, input), origin);
|
|
26
|
+
if (url.origin !== new URL(origin).origin)
|
|
27
|
+
throw new Error('plan URL escaped its origin');
|
|
28
|
+
if (at !== -1) {
|
|
29
|
+
for (const [key, value] of new URLSearchParams(template.slice(at + 1)))
|
|
30
|
+
url.searchParams.append(key, render(value, input));
|
|
31
|
+
}
|
|
32
|
+
return url.toString();
|
|
33
|
+
}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import { mkdtemp, rm, writeFile } from 'node:fs/promises';
|
|
2
|
+
import { readFileSync } from 'node:fs';
|
|
3
|
+
import { tmpdir } from 'node:os';
|
|
4
|
+
import { join } from 'node:path';
|
|
5
|
+
import { createHash } from 'node:crypto';
|
|
6
|
+
/** Bodies at or below this stay in the trace; larger ones spill to a file. */
|
|
7
|
+
const INLINE_LIMIT_BYTES = 256 * 1024;
|
|
8
|
+
/** A body this large is a mistake, not a page. Recorded as truncated. */
|
|
9
|
+
const SAFETY_CAP_BYTES = 64 * 1024 * 1024;
|
|
10
|
+
/**
|
|
11
|
+
* Preserves response bodies of any size.
|
|
12
|
+
*
|
|
13
|
+
* A fixed inline cap silently dropped flathub.org's 662KB search page, which
|
|
14
|
+
* left the html compiler with nothing to work from — the strategy was not
|
|
15
|
+
* refused, it was absent. Size is a property of the page, not a reason to lose
|
|
16
|
+
* the observation, so large bodies go to a file and the trace keeps a pointer.
|
|
17
|
+
* That also keeps a serialised trace small enough to read.
|
|
18
|
+
*/
|
|
19
|
+
export class BodyStore {
|
|
20
|
+
dir;
|
|
21
|
+
constructor(dir) {
|
|
22
|
+
this.dir = dir;
|
|
23
|
+
}
|
|
24
|
+
static async create() {
|
|
25
|
+
return new BodyStore(await mkdtemp(join(tmpdir(), 'fwa-bodies-')));
|
|
26
|
+
}
|
|
27
|
+
async put(key, buffer, cap = SAFETY_CAP_BYTES) {
|
|
28
|
+
const bodySize = buffer.byteLength;
|
|
29
|
+
if (bodySize > cap) {
|
|
30
|
+
return { body: null, bodyPath: null, bodySize, truncated: true };
|
|
31
|
+
}
|
|
32
|
+
if (bodySize <= INLINE_LIMIT_BYTES) {
|
|
33
|
+
return { body: buffer.toString('utf8'), bodyPath: null, bodySize, truncated: false };
|
|
34
|
+
}
|
|
35
|
+
const name = createHash('sha256').update(key).digest('hex').slice(0, 16);
|
|
36
|
+
const bodyPath = join(this.dir, `${name}.body`);
|
|
37
|
+
await writeFile(bodyPath, buffer);
|
|
38
|
+
return { body: null, bodyPath, bodySize, truncated: false };
|
|
39
|
+
}
|
|
40
|
+
async dispose() {
|
|
41
|
+
await rm(this.dir, { recursive: true, force: true });
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* The body of a recorded request, wherever it was kept. Synchronous because
|
|
46
|
+
* every caller is a compiler reading one trace, not a hot path.
|
|
47
|
+
*/
|
|
48
|
+
export function readBody(request) {
|
|
49
|
+
if (request.body !== null)
|
|
50
|
+
return request.body;
|
|
51
|
+
if (request.bodyPath === null)
|
|
52
|
+
return null;
|
|
53
|
+
try {
|
|
54
|
+
return readFileSync(request.bodyPath, 'utf8');
|
|
55
|
+
}
|
|
56
|
+
catch {
|
|
57
|
+
return null;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
import { openSession } from '../browser/session.js';
|
|
2
|
+
import { BodyStore } from './body.js';
|
|
3
|
+
import { navigateAndSettle } from '../browser/navigate.js';
|
|
4
|
+
/** A DOM mutation this soon after a response is treated as caused by it. */
|
|
5
|
+
const DOM_SETTLE_MS = 800;
|
|
6
|
+
/**
|
|
7
|
+
* Records DOM mutations and request completions on the *page's* clock.
|
|
8
|
+
*
|
|
9
|
+
* Both must come from inside the page. A response event observed in Node and a
|
|
10
|
+
* mutation observed in the page are recorded at different stages of the
|
|
11
|
+
* pipeline, so under load their order can invert and a later request steals
|
|
12
|
+
* credit for an earlier response's DOM change. Inside the page the ordering is
|
|
13
|
+
* guaranteed by the event loop: the mutation callback is a microtask that runs
|
|
14
|
+
* before any subsequent `await fetch(...)` resolves.
|
|
15
|
+
*
|
|
16
|
+
* Observes `document`, not `document.documentElement`: init scripts run at
|
|
17
|
+
* document-start, where documentElement is still null and observing it throws.
|
|
18
|
+
*/
|
|
19
|
+
const PAGE_PROBE = `
|
|
20
|
+
window.__fwaMutations = []
|
|
21
|
+
new MutationObserver(() => { window.__fwaMutations.push(Date.now()) })
|
|
22
|
+
.observe(document, { childList: true, subtree: true, characterData: true })
|
|
23
|
+
|
|
24
|
+
window.__fwaCompletions = []
|
|
25
|
+
const __fwaRecord = (raw) => {
|
|
26
|
+
try {
|
|
27
|
+
window.__fwaCompletions.push({ url: new URL(raw, location.href).href, at: Date.now() })
|
|
28
|
+
} catch {}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const __fwaFetch = window.fetch
|
|
32
|
+
window.fetch = async function (...args) {
|
|
33
|
+
const response = await __fwaFetch.apply(this, args)
|
|
34
|
+
const first = args[0]
|
|
35
|
+
__fwaRecord(typeof first === 'string' ? first : (first && first.url) || String(first))
|
|
36
|
+
return response
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const __fwaOpen = XMLHttpRequest.prototype.open
|
|
40
|
+
const __fwaSend = XMLHttpRequest.prototype.send
|
|
41
|
+
XMLHttpRequest.prototype.open = function (method, url, ...rest) {
|
|
42
|
+
this.__fwaUrl = url
|
|
43
|
+
return __fwaOpen.call(this, method, url, ...rest)
|
|
44
|
+
}
|
|
45
|
+
XMLHttpRequest.prototype.send = function (...args) {
|
|
46
|
+
this.addEventListener('loadend', () => { __fwaRecord(this.__fwaUrl) })
|
|
47
|
+
return __fwaSend.apply(this, args)
|
|
48
|
+
}
|
|
49
|
+
`;
|
|
50
|
+
/**
|
|
51
|
+
* Replaces each request's Node-observed timestamp with the page-observed one
|
|
52
|
+
* where the page saw it, so that mutations and completions share a clock.
|
|
53
|
+
*/
|
|
54
|
+
export function alignToPageClock(requests, completions) {
|
|
55
|
+
const unused = [...completions];
|
|
56
|
+
for (const request of requests) {
|
|
57
|
+
const idx = unused.findIndex((c) => c.url === request.url);
|
|
58
|
+
if (idx === -1)
|
|
59
|
+
continue;
|
|
60
|
+
request.at = unused[idx].at;
|
|
61
|
+
unused.splice(idx, 1);
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Credits each DOM mutation to the single most recent response that preceded
|
|
66
|
+
* it. Marking every response inside a time window instead would be useless on
|
|
67
|
+
* a fast site, where a decoy landing 1ms before the real response would be
|
|
68
|
+
* credited with the same mutation.
|
|
69
|
+
*
|
|
70
|
+
* `requests` must be in arrival order, which is how the response handler builds it.
|
|
71
|
+
*/
|
|
72
|
+
export function attributeMutations(requests, mutations) {
|
|
73
|
+
for (const at of mutations) {
|
|
74
|
+
let cause = null;
|
|
75
|
+
for (const request of requests) {
|
|
76
|
+
if (request.at <= at && at - request.at <= DOM_SETTLE_MS)
|
|
77
|
+
cause = request;
|
|
78
|
+
}
|
|
79
|
+
if (cause)
|
|
80
|
+
cause.domChanged = true;
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* A trace owns the files its large bodies were spilled into, so a caller that
|
|
85
|
+
* keeps the trace past the call must dispose of it.
|
|
86
|
+
*/
|
|
87
|
+
export async function record(plan, task, sites) {
|
|
88
|
+
const origin = sites.origin(task.site);
|
|
89
|
+
const session = await openSession();
|
|
90
|
+
const actions = [];
|
|
91
|
+
const pending = [];
|
|
92
|
+
const bodyReads = [];
|
|
93
|
+
const bodies = await BodyStore.create();
|
|
94
|
+
try {
|
|
95
|
+
await session.page.addInitScript(PAGE_PROBE);
|
|
96
|
+
session.page.on('response', (response) => {
|
|
97
|
+
const request = response.request();
|
|
98
|
+
const contentType = response.headers()['content-type'] ?? null;
|
|
99
|
+
const entry = {
|
|
100
|
+
method: request.method(),
|
|
101
|
+
url: response.url(),
|
|
102
|
+
resourceType: request.resourceType(),
|
|
103
|
+
status: response.status(),
|
|
104
|
+
contentType,
|
|
105
|
+
body: null,
|
|
106
|
+
bodyPath: null,
|
|
107
|
+
bodySize: 0,
|
|
108
|
+
truncated: false,
|
|
109
|
+
postData: request.postData(),
|
|
110
|
+
requestHeaders: request.headers(),
|
|
111
|
+
at: Date.now(),
|
|
112
|
+
afterAction: actions.length === 0 ? null : actions.length - 1,
|
|
113
|
+
domChanged: false,
|
|
114
|
+
};
|
|
115
|
+
pending.push(entry);
|
|
116
|
+
if (contentType && /json|text\/plain|text\/html/.test(contentType)) {
|
|
117
|
+
bodyReads.push(response
|
|
118
|
+
.body()
|
|
119
|
+
.then(async (buf) => {
|
|
120
|
+
Object.assign(entry, await bodies.put(`${entry.method} ${entry.url} ${entry.at}`, buf));
|
|
121
|
+
})
|
|
122
|
+
.catch(() => undefined));
|
|
123
|
+
}
|
|
124
|
+
});
|
|
125
|
+
const target = plan.url(origin, task);
|
|
126
|
+
actions.push({ index: 0, type: 'navigate', value: target, at: Date.now() });
|
|
127
|
+
await navigateAndSettle(session.page, target, plan.itemSelector);
|
|
128
|
+
const mutations = (await session.page.evaluate('window.__fwaMutations || []'));
|
|
129
|
+
const completions = (await session.page.evaluate('window.__fwaCompletions || []'));
|
|
130
|
+
// Align first: attribution compares these timestamps against each other.
|
|
131
|
+
alignToPageClock(pending, completions);
|
|
132
|
+
attributeMutations(pending, mutations);
|
|
133
|
+
const finalHtml = await session.page.content();
|
|
134
|
+
await Promise.all(bodyReads);
|
|
135
|
+
return {
|
|
136
|
+
site: task.site,
|
|
137
|
+
intent: task.intent,
|
|
138
|
+
input: task.input,
|
|
139
|
+
origin,
|
|
140
|
+
actions,
|
|
141
|
+
requests: pending,
|
|
142
|
+
finalHtml,
|
|
143
|
+
...(session.page.viewportSize() === null ? {} : { viewport: session.page.viewportSize() }),
|
|
144
|
+
cost: session.cost,
|
|
145
|
+
dispose: () => bodies.dispose(),
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
finally {
|
|
149
|
+
await session.close();
|
|
150
|
+
}
|
|
151
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
export class StaticSiteResolver {
|
|
2
|
+
map;
|
|
3
|
+
constructor(map) {
|
|
4
|
+
this.map = map;
|
|
5
|
+
}
|
|
6
|
+
origin(site) {
|
|
7
|
+
const origin = this.map[site];
|
|
8
|
+
if (!origin)
|
|
9
|
+
throw new Error(`no origin registered for site "${site}"`);
|
|
10
|
+
return origin;
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
/** Origins for the wild benchmark set. Fixture origins are supplied per-run. */
|
|
14
|
+
export const WILD_ORIGINS = {
|
|
15
|
+
'pkg.go.dev': 'https://pkg.go.dev',
|
|
16
|
+
'crates.io': 'https://crates.io',
|
|
17
|
+
'hn.algolia.com': 'https://hn.algolia.com',
|
|
18
|
+
// Held-out set: chosen after the engine was written, and run without tuning.
|
|
19
|
+
'hex.pm': 'https://hex.pm',
|
|
20
|
+
'docs.rs': 'https://docs.rs',
|
|
21
|
+
'flathub.org': 'https://flathub.org',
|
|
22
|
+
'jsr.io': 'https://jsr.io',
|
|
23
|
+
// Held-out 2: chosen for families the engine has never seen — a forum, two
|
|
24
|
+
// job boards, a marketplace, and a shadow-DOM SPA.
|
|
25
|
+
'tildes.net': 'https://tildes.net',
|
|
26
|
+
'remoteok.com': 'https://remoteok.com',
|
|
27
|
+
'itch.io': 'https://itch.io',
|
|
28
|
+
'arbeitnow.com': 'https://www.arbeitnow.com',
|
|
29
|
+
'archive.org': 'https://archive.org',
|
|
30
|
+
// Held-out 3: drawn under a frozen pre-registration, four challenge slots and
|
|
31
|
+
// one positive control. Every one of them disallows /search, so each task
|
|
32
|
+
// addresses the site's own browse axis instead.
|
|
33
|
+
'meta.discourse.org': 'https://meta.discourse.org',
|
|
34
|
+
'loc.gov': 'https://www.loc.gov',
|
|
35
|
+
'bandcamp.com': 'https://bandcamp.com',
|
|
36
|
+
'lemmy.world': 'https://lemmy.world',
|
|
37
|
+
'openlibrary.org': 'https://openlibrary.org',
|
|
38
|
+
// Held-out 4: two fediverse apps of different software drawn for cursor
|
|
39
|
+
// pagination, a metadata catalog, a client-heavy forum, and a control.
|
|
40
|
+
'mastodon.social': 'https://mastodon.social',
|
|
41
|
+
'pixelfed.social': 'https://pixelfed.social',
|
|
42
|
+
'musicbrainz.org': 'https://musicbrainz.org',
|
|
43
|
+
'dev.to': 'https://dev.to',
|
|
44
|
+
'npmjs.com': 'https://www.npmjs.com',
|
|
45
|
+
};
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { localPaths, verifyReadable, verificationWarnings, UserError } from './local.js';
|
|
2
|
+
import { buildEngine } from './wiring.js';
|
|
3
|
+
import { loadLearnedPlans, mergePlans } from './authoring/plans.js';
|
|
4
|
+
import { PLANS } from '../benchmark/plans.js';
|
|
5
|
+
import { WILD_ORIGINS } from './sites.js';
|
|
6
|
+
/** The hand-written plans, with anything saved layered over them. */
|
|
7
|
+
export async function allPlans(planDir) {
|
|
8
|
+
const learned = await loadLearnedPlans(planDir);
|
|
9
|
+
return { plans: mergePlans(PLANS, learned.plans), origins: learned.origins, verifications: learned.verifications };
|
|
10
|
+
}
|
|
11
|
+
/** Every saved task, as `site/intent`, with its storage directory. */
|
|
12
|
+
export async function listTasks(dataDir) {
|
|
13
|
+
const paths = localPaths(dataDir);
|
|
14
|
+
const { plans } = await loadLearnedPlans(paths.plans);
|
|
15
|
+
const tasks = Object.entries(plans).flatMap(([site, intents]) => Object.keys(intents).map((intent) => `${site}/${intent}`));
|
|
16
|
+
return { root: paths.root, tasks };
|
|
17
|
+
}
|
|
18
|
+
export async function fetchTask(site, intent, input, opts = {}) {
|
|
19
|
+
const paths = localPaths(opts.dataDir);
|
|
20
|
+
const { plans, origins, verifications } = await allPlans(paths.plans);
|
|
21
|
+
const plan = plans[site]?.[intent];
|
|
22
|
+
if (!plan)
|
|
23
|
+
throw new UserError('NOT_TAUGHT', `No recipe saved as ${site}/${intent}. Run inspect on the page, then save.`);
|
|
24
|
+
// Reject missing placeholders before starting a browser or making a request.
|
|
25
|
+
plan.url(origins[site] ?? WILD_ORIGINS[site], { id: 'check', site, intent, input });
|
|
26
|
+
const engine = buildEngine({ recipeDir: paths.recipes, plans, origins, heal: opts.heal });
|
|
27
|
+
try {
|
|
28
|
+
const outcome = await engine.executor.run({ id: opts.runId ?? 'fetch', site, intent, input });
|
|
29
|
+
const verification = verifyReadable(outcome.items, Object.keys(plan.fields), verifications[site]?.[intent]);
|
|
30
|
+
// Derived from the checks, so a run that proves more says less.
|
|
31
|
+
const warnings = [...outcome.reasons, ...verificationWarnings(verification)];
|
|
32
|
+
return { outcome, verification, warnings };
|
|
33
|
+
}
|
|
34
|
+
finally {
|
|
35
|
+
await engine.warm.close();
|
|
36
|
+
}
|
|
37
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a task cost in total, not what the attempt that happened to answer it
|
|
3
|
+
* cost. A run that tried a recipe, lost, fell back to a browser and re-recorded
|
|
4
|
+
* on the way paid for all three; reporting only the last describes a cheaper
|
|
5
|
+
* run than the one that happened.
|
|
6
|
+
*
|
|
7
|
+
* `answered` is the attempt that produced the items, so it keeps the strategy
|
|
8
|
+
* name and the token count: the agent reads one answer, not three.
|
|
9
|
+
*/
|
|
10
|
+
export function addMeta(answered, ...spent) {
|
|
11
|
+
return spent.reduce((total, one) => ({
|
|
12
|
+
...total,
|
|
13
|
+
latencyMs: total.latencyMs + one.latencyMs,
|
|
14
|
+
browserLaunches: total.browserLaunches + one.browserLaunches,
|
|
15
|
+
pageNavigations: total.pageNavigations + one.pageNavigations,
|
|
16
|
+
networkRequests: total.networkRequests + one.networkRequests,
|
|
17
|
+
bytesDownloaded: total.bytesDownloaded + one.bytesDownloaded,
|
|
18
|
+
politenessWaitMs: total.politenessWaitMs + one.politenessWaitMs,
|
|
19
|
+
}), answered);
|
|
20
|
+
}
|
|
21
|
+
export function emptyMeta(strategy) {
|
|
22
|
+
return {
|
|
23
|
+
strategy,
|
|
24
|
+
latencyMs: 0,
|
|
25
|
+
browserLaunches: 0,
|
|
26
|
+
pageNavigations: 0,
|
|
27
|
+
networkRequests: 0,
|
|
28
|
+
bytesDownloaded: 0,
|
|
29
|
+
llmTokens: 0,
|
|
30
|
+
politenessWaitMs: 0,
|
|
31
|
+
};
|
|
32
|
+
}
|