webrecipe 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -15
- package/dist/src/browser/navigate.js +101 -7
- package/dist/src/executor/index.js +6 -0
- package/dist/src/executor/strategies/browser.js +8 -3
- package/dist/src/executor/strategies/warm-browser.js +8 -3
- package/dist/src/healing/index.js +1 -1
- package/dist/src/local.js +8 -0
- package/dist/src/net/politeness.js +11 -2
- package/dist/src/recorder/index.js +5 -3
- package/dist/src/tasks.js +20 -3
- package/dist/src/wiring.js +5 -3
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -18,9 +18,8 @@ search results. Runs locally. No hosted account, no API key.
|
|
|
18
18
|
```
|
|
19
19
|
$ webrecipe fetch hn/list
|
|
20
20
|
title url
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
Show HN: Xyp RSS – hold to play, swipe to skip, highlight a word for graph https://xyp.app
|
|
21
|
+
Special agents' blood and urine test results… https://www.bbc.co.uk/news/...
|
|
22
|
+
Show HN: Xyp RSS – hold to play, swipe to skip… https://xyp.app
|
|
24
23
|
...
|
|
25
24
|
```
|
|
26
25
|
|
|
@@ -55,7 +54,7 @@ webrecipe fetch hn/list # TSV on stdout
|
|
|
55
54
|
webrecipe fetch hn/list --json # one JSON object on stdout
|
|
56
55
|
```
|
|
57
56
|
|
|
58
|
-
Run it again tomorrow and you get tomorrow's
|
|
57
|
+
Run it again tomorrow and you get tomorrow's latest submissions.
|
|
59
58
|
|
|
60
59
|
## Use with an agent
|
|
61
60
|
|
|
@@ -128,10 +127,20 @@ How to choose:
|
|
|
128
127
|
- Scores do not tell you meaning. Above, the vote link scores as well as the
|
|
129
128
|
story link. Read the sample column to tell them apart.
|
|
130
129
|
|
|
131
|
-
**2. Save** your choice
|
|
130
|
+
**2. Save** your choice. Replace `NAME` with the output field name you want,
|
|
131
|
+
such as `title` or `url`.
|
|
132
|
+
|
|
133
|
+
```sh
|
|
134
|
+
webrecipe save hn/list --url https://news.ycombinator.com/newest \
|
|
135
|
+
--items 'tr.athing.submission' \
|
|
136
|
+
--field 'title=span.titleline > a' 'url=span.titleline > a@href'
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
The name is `site/intent`. `site` is any name you choose, not necessarily the
|
|
140
|
+
domain, so `hn`, `hn-jobs` and `my-shop` are all fine. `intent` is `list`,
|
|
132
141
|
`search` or `detail`. Saving the same name again replaces it.
|
|
133
142
|
|
|
134
|
-
**3. Fetch** by that name
|
|
143
|
+
**3. Fetch** by that name: `webrecipe fetch hn/list`.
|
|
135
144
|
|
|
136
145
|
### Pages with a parameter
|
|
137
146
|
|
|
@@ -197,6 +206,7 @@ A failure prints one object and exits 1:
|
|
|
197
206
|
| --- | --- |
|
|
198
207
|
| `NOT_TAUGHT` | nothing saved under that `site/intent` |
|
|
199
208
|
| `INVALID_INPUT` | bad arguments, or an input the recipe does not take |
|
|
209
|
+
| `ROBOTS_DISALLOWED` | robots.txt disallows the page or a redirect target; the disallowed URL was not requested |
|
|
200
210
|
| `UNVERIFIED_RESULT` | zero rows, or a selected field missing from some rows |
|
|
201
211
|
| `EXECUTION_FAILED` | network, browser or storage error |
|
|
202
212
|
|
|
@@ -226,8 +236,13 @@ Every `inspect`, `save`, `fetch` and `read` appends one line to a local log.
|
|
|
226
236
|
timing and outcome, never page content, cookies or headers. Nothing is
|
|
227
237
|
uploaded anywhere. `--no-log` skips logging for one command.
|
|
228
238
|
|
|
229
|
-
|
|
230
|
-
|
|
239
|
+
HTTP requests identify themselves with a `webrecipe/0.1` user agent and wait
|
|
240
|
+
between requests to the same host, honouring robots.txt `Crawl-delay`.
|
|
241
|
+
`fetch` refuses a URL that robots.txt disallows before requesting it, and
|
|
242
|
+
checks every redirect target the same way, over HTTP and in the browser.
|
|
243
|
+
`read` refuses a disallowed URL before requesting it. `inspect` and `save` load the page in a
|
|
244
|
+
browser without checking robots.txt. Check a site's terms before saving a
|
|
245
|
+
recipe for it.
|
|
231
246
|
|
|
232
247
|
## Also: `read`
|
|
233
248
|
|
|
@@ -259,9 +274,11 @@ politeness waits and Node startup
|
|
|
259
274
|
| Steam store page | 0.3 s | 2.9 s | 162 KB | 33.6 MB |
|
|
260
275
|
|
|
261
276
|
**The first read is not free.** Inspect plus save took 5.4 s on Hacker News,
|
|
262
|
-
52 s on Remote OK and 15 s on Steam
|
|
263
|
-
selectors
|
|
264
|
-
|
|
277
|
+
52 s on Remote OK and 15 s on Steam. Those figures exclude the time to choose
|
|
278
|
+
selectors and any politeness wait; Hacker News added 27 s of waiting between
|
|
279
|
+
inspect and save. Counting setup and waits in cumulative wall time, repeat fetches
|
|
280
|
+
overtook the browser at repetition 2, 18 and 6 respectively. If you will read
|
|
281
|
+
a page once, use `read`.
|
|
265
282
|
|
|
266
283
|
Hacker News asks crawlers to wait 30 s between page loads. With that wait
|
|
267
284
|
included, both approaches are dominated by it: a median 28.0 s per fetch for
|
|
@@ -275,10 +292,11 @@ hand. But of the 23 answers checked by hand, 12 had reached `verified` on the
|
|
|
275
292
|
structural checks alone. That is where a wrong answer would hide, and one
|
|
276
293
|
field there was ambiguous in exactly that way.
|
|
277
294
|
|
|
278
|
-
**Sites drift.**
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
295
|
+
**Sites drift.** A page that saved cleanly one day failed to save the next,
|
|
296
|
+
because a required field was missing ([notes](benchmark/results/drift/)).
|
|
297
|
+
Separate requests that day also got a human-verification page. The cause of
|
|
298
|
+
the original failure was not established. The save failed loudly rather than
|
|
299
|
+
storing a recipe with an empty field.
|
|
282
300
|
|
|
283
301
|
## Development
|
|
284
302
|
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { UserError } from '../local.js';
|
|
1
2
|
const NAVIGATION_TIMEOUT_MS = 30_000;
|
|
2
3
|
const SELECTOR_TIMEOUT_MS = 10_000;
|
|
3
4
|
/** After the items appear, give late XHR a moment to land before reading. */
|
|
@@ -12,11 +13,104 @@ export const SETTLE_MS = 800;
|
|
|
12
13
|
* the selector, with a short settle for anything still in flight, and a
|
|
13
14
|
* network-idle attempt only as a best effort that is allowed to fail.
|
|
14
15
|
*/
|
|
15
|
-
export async function navigateAndSettle(page, url, itemSelector) {
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
.
|
|
20
|
-
|
|
21
|
-
|
|
16
|
+
export async function navigateAndSettle(page, url, itemSelector, guard) {
|
|
17
|
+
if (guard)
|
|
18
|
+
await guard.goto(url);
|
|
19
|
+
else
|
|
20
|
+
await page.goto(url, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS });
|
|
21
|
+
// Waited for again while the page is still moving: the items worth reading
|
|
22
|
+
// are the ones on the page it ends on, not the one it left.
|
|
23
|
+
for (let moves = 0;; moves++) {
|
|
24
|
+
await page
|
|
25
|
+
.waitForSelector(itemSelector, { timeout: SELECTOR_TIMEOUT_MS })
|
|
26
|
+
.catch(() => undefined);
|
|
27
|
+
await page.waitForLoadState('networkidle', { timeout: SETTLE_MS }).catch(() => undefined);
|
|
28
|
+
await page.waitForTimeout(SETTLE_MS);
|
|
29
|
+
if (!guard || moves >= MAX_REDIRECTS || !(await guard.arrived()))
|
|
30
|
+
break;
|
|
31
|
+
}
|
|
32
|
+
guard?.check();
|
|
33
|
+
}
|
|
34
|
+
const MAX_REDIRECTS = 20;
|
|
35
|
+
/**
|
|
36
|
+
* Checks every main-frame navigation of a page, for as long as the page lives.
|
|
37
|
+
*
|
|
38
|
+
* A route handler alone cannot check redirects: Chromium follows a redirect
|
|
39
|
+
* without routing the next hop, so the disallowed page would be requested
|
|
40
|
+
* before any handler saw it. Instead the handler fetches each main-frame
|
|
41
|
+
* document without following redirects, stops the navigation at a redirect,
|
|
42
|
+
* and starts a fresh one to the target, which passes through the check again.
|
|
43
|
+
*
|
|
44
|
+
* This happens for the page's whole life, not only inside `goto`: a page can
|
|
45
|
+
* move itself after loading, and a redirect then has to be followed, or
|
|
46
|
+
* refused, rather than dropped with the old page left to be read.
|
|
47
|
+
*/
|
|
48
|
+
export async function guardPage(page, allowed) {
|
|
49
|
+
let refused = null;
|
|
50
|
+
let failure = null;
|
|
51
|
+
let hops = 0;
|
|
52
|
+
const following = new Set();
|
|
53
|
+
const refusal = () => new UserError('ROBOTS_DISALLOWED', `robots.txt disallows ${refused}`);
|
|
54
|
+
const check = () => {
|
|
55
|
+
if (refused !== null)
|
|
56
|
+
throw refusal();
|
|
57
|
+
if (failure !== null)
|
|
58
|
+
throw failure;
|
|
59
|
+
};
|
|
60
|
+
const follow = (target) => {
|
|
61
|
+
if (++hops > MAX_REDIRECTS) {
|
|
62
|
+
failure ??= new Error('too many redirects');
|
|
63
|
+
return;
|
|
64
|
+
}
|
|
65
|
+
const moving = page.goto(target, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS })
|
|
66
|
+
.then(() => undefined, () => undefined)
|
|
67
|
+
.finally(() => { following.delete(moving); });
|
|
68
|
+
following.add(moving);
|
|
69
|
+
};
|
|
70
|
+
await page.route('**/*', async (route) => {
|
|
71
|
+
try {
|
|
72
|
+
const request = route.request();
|
|
73
|
+
if (!request.isNavigationRequest() || request.frame() !== page.mainFrame())
|
|
74
|
+
return await route.continue();
|
|
75
|
+
if (!(await allowed(request.url()))) {
|
|
76
|
+
refused ??= request.url();
|
|
77
|
+
return await route.abort('blockedbyclient');
|
|
78
|
+
}
|
|
79
|
+
const response = await route.fetch({ maxRedirects: 0 });
|
|
80
|
+
const location = response.headers()['location'];
|
|
81
|
+
if (response.status() >= 300 && response.status() < 400 && location) {
|
|
82
|
+
await route.abort('aborted');
|
|
83
|
+
follow(new URL(location, request.url()).toString());
|
|
84
|
+
return;
|
|
85
|
+
}
|
|
86
|
+
await route.fulfill({ response });
|
|
87
|
+
}
|
|
88
|
+
catch {
|
|
89
|
+
// The page closed under a pending request; there is no one left to answer.
|
|
90
|
+
}
|
|
91
|
+
});
|
|
92
|
+
let seen = 0;
|
|
93
|
+
const arrived = async () => {
|
|
94
|
+
while (following.size > 0)
|
|
95
|
+
await Promise.all(following);
|
|
96
|
+
check();
|
|
97
|
+
const moved = hops !== seen;
|
|
98
|
+
seen = hops;
|
|
99
|
+
return moved;
|
|
100
|
+
};
|
|
101
|
+
return {
|
|
102
|
+
async goto(url) {
|
|
103
|
+
// A navigation stopped at a redirect rejects; what happens next is `follow`'s, and `arrived` waits for it.
|
|
104
|
+
try {
|
|
105
|
+
await page.goto(url, { waitUntil: 'domcontentloaded', timeout: NAVIGATION_TIMEOUT_MS });
|
|
106
|
+
}
|
|
107
|
+
catch (error) {
|
|
108
|
+
if (refused === null && hops === 0)
|
|
109
|
+
throw error;
|
|
110
|
+
}
|
|
111
|
+
await arrived();
|
|
112
|
+
},
|
|
113
|
+
arrived,
|
|
114
|
+
check,
|
|
115
|
+
};
|
|
22
116
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { measureResult } from '../measurement.js';
|
|
2
2
|
import { validate } from '../validator/index.js';
|
|
3
|
+
import { isRobotsRefusal } from '../local.js';
|
|
3
4
|
import { addMeta } from '../types.js';
|
|
4
5
|
/** How long a site stays off the recipe path after it refused one. */
|
|
5
6
|
export const BLOCK_COOLDOWN_MS = 10 * 60_000;
|
|
@@ -45,6 +46,8 @@ export class Executor {
|
|
|
45
46
|
return { ...result, recipeUsed: false, fellBack: true, reasons };
|
|
46
47
|
}
|
|
47
48
|
catch (err) {
|
|
49
|
+
if (isRobotsRefusal(err))
|
|
50
|
+
throw err;
|
|
48
51
|
lastError = err;
|
|
49
52
|
reasons.push(err instanceof Error ? err.message : String(err));
|
|
50
53
|
}
|
|
@@ -92,6 +95,9 @@ export class Executor {
|
|
|
92
95
|
}
|
|
93
96
|
}
|
|
94
97
|
catch (err) {
|
|
98
|
+
// Not a reason to try the browser: it would request the page robots.txt excluded.
|
|
99
|
+
if (isRobotsRefusal(err))
|
|
100
|
+
throw err;
|
|
95
101
|
reasons.push(err instanceof Error ? err.message : String(err));
|
|
96
102
|
}
|
|
97
103
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { measureResult } from '../../measurement.js';
|
|
2
2
|
import { openSession } from '../../browser/session.js';
|
|
3
|
-
import { navigateAndSettle } from '../../browser/navigate.js';
|
|
3
|
+
import { navigateAndSettle, guardPage } from '../../browser/navigate.js';
|
|
4
4
|
import { planFields } from '../extract.js';
|
|
5
5
|
import { countTokens } from '../tokens.js';
|
|
6
6
|
import { emptyMeta } from '../../types.js';
|
|
@@ -9,13 +9,15 @@ export class BrowserStrategy {
|
|
|
9
9
|
sites;
|
|
10
10
|
plans;
|
|
11
11
|
countAgentTokens;
|
|
12
|
+
guard;
|
|
12
13
|
name = 'browser';
|
|
13
14
|
constructor(sites, plans = BROWSER_PLANS,
|
|
14
15
|
/** The token count costs a page read of its own; a timing benchmark can decline to pay it. */
|
|
15
|
-
countAgentTokens = true) {
|
|
16
|
+
countAgentTokens = true, guard) {
|
|
16
17
|
this.sites = sites;
|
|
17
18
|
this.plans = plans;
|
|
18
19
|
this.countAgentTokens = countAgentTokens;
|
|
20
|
+
this.guard = guard;
|
|
19
21
|
}
|
|
20
22
|
async execute(recipe, task) {
|
|
21
23
|
return measureResult(this.name, () => this.executeAttempt(recipe, task));
|
|
@@ -29,7 +31,8 @@ export class BrowserStrategy {
|
|
|
29
31
|
const session = await openSession();
|
|
30
32
|
meta.browserLaunches = 1;
|
|
31
33
|
try {
|
|
32
|
-
await
|
|
34
|
+
const guard = this.guard ? await guardPage(session.page, this.guard) : undefined;
|
|
35
|
+
await navigateAndSettle(session.page, plan.url(this.sites.origin(task.site), task), plan.itemSelector, guard);
|
|
33
36
|
// The field specs are interpreted once, in node, so that the browser and
|
|
34
37
|
// cheerio cannot drift apart on what a spec means.
|
|
35
38
|
const items = (await session.page.$$eval(plan.itemSelector, (elements, plans) => elements.map((el) => Object.fromEntries(plans.map((f) => {
|
|
@@ -51,6 +54,8 @@ export class BrowserStrategy {
|
|
|
51
54
|
// graded run into a thrown one.
|
|
52
55
|
const snapshot = this.countAgentTokens ? await session.page.locator('body').ariaSnapshot().catch(() => '') : '';
|
|
53
56
|
meta.llmTokens = countTokens(snapshot);
|
|
57
|
+
// After reading, so a page that moved somewhere disallowed while it was read is not answered from.
|
|
58
|
+
guard?.check();
|
|
54
59
|
return { items, meta };
|
|
55
60
|
}
|
|
56
61
|
finally {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { measureResult } from '../../measurement.js';
|
|
2
2
|
import { BrowserPool } from '../../browser/pool.js';
|
|
3
|
-
import { navigateAndSettle } from '../../browser/navigate.js';
|
|
3
|
+
import { navigateAndSettle, guardPage } from '../../browser/navigate.js';
|
|
4
4
|
import { planFields } from '../extract.js';
|
|
5
5
|
import { countTokens } from '../tokens.js';
|
|
6
6
|
import { BROWSER_PLANS } from './browser.js';
|
|
@@ -9,11 +9,13 @@ import { emptyMeta } from '../../types.js';
|
|
|
9
9
|
export class WarmBrowserStrategy {
|
|
10
10
|
sites;
|
|
11
11
|
plans;
|
|
12
|
+
guard;
|
|
12
13
|
name = 'warm-browser';
|
|
13
14
|
pool = new BrowserPool();
|
|
14
|
-
constructor(sites, plans = BROWSER_PLANS) {
|
|
15
|
+
constructor(sites, plans = BROWSER_PLANS, guard) {
|
|
15
16
|
this.sites = sites;
|
|
16
17
|
this.plans = plans;
|
|
18
|
+
this.guard = guard;
|
|
17
19
|
}
|
|
18
20
|
async execute(recipe, task) {
|
|
19
21
|
return measureResult(this.name, () => this.executeAttempt(recipe, task));
|
|
@@ -26,7 +28,8 @@ export class WarmBrowserStrategy {
|
|
|
26
28
|
const started = performance.now();
|
|
27
29
|
const warm = await this.pool.acquire();
|
|
28
30
|
try {
|
|
29
|
-
await
|
|
31
|
+
const guard = this.guard ? await guardPage(warm.page, this.guard) : undefined;
|
|
32
|
+
await navigateAndSettle(warm.page, plan.url(this.sites.origin(task.site), task), plan.itemSelector, guard);
|
|
30
33
|
const items = (await warm.page.$$eval(plan.itemSelector, (elements, plans) => elements.map((el) => Object.fromEntries(plans.map((f) => {
|
|
31
34
|
switch (f.mode) {
|
|
32
35
|
case 'own-text': return [f.name, el.textContent?.trim() ?? null];
|
|
@@ -45,6 +48,8 @@ export class WarmBrowserStrategy {
|
|
|
45
48
|
// After the latency line, and allowed to fail, for the reasons in browser.ts.
|
|
46
49
|
const snapshot = await warm.page.locator('body').ariaSnapshot().catch(() => '');
|
|
47
50
|
meta.llmTokens = countTokens(snapshot);
|
|
51
|
+
// After reading, so a page that moved somewhere disallowed while it was read is not answered from.
|
|
52
|
+
guard?.check();
|
|
48
53
|
return { items, meta };
|
|
49
54
|
}
|
|
50
55
|
finally {
|
|
@@ -70,7 +70,7 @@ export class SelfHealer {
|
|
|
70
70
|
if (!plan)
|
|
71
71
|
return { healed: false, recipe: null, changes: [], refused: null, meta: null };
|
|
72
72
|
const started = performance.now();
|
|
73
|
-
const trace = await record(plan, event.task, this.opts.sites);
|
|
73
|
+
const trace = await record(plan, event.task, this.opts.sites, this.opts.guard);
|
|
74
74
|
try {
|
|
75
75
|
const meta = costOf(trace, Math.round(performance.now() - started));
|
|
76
76
|
// Every compiler that ran and refused, not only the last. The html one
|
package/dist/src/local.js
CHANGED
|
@@ -109,6 +109,14 @@ export class UserError extends Error {
|
|
|
109
109
|
this.code = code;
|
|
110
110
|
}
|
|
111
111
|
}
|
|
112
|
+
/** A robots.txt refusal ends the task: no fallback, no retry on another path, no relearn. */
|
|
113
|
+
export function isRobotsRefusal(error) {
|
|
114
|
+
for (let e = error; e instanceof Error; e = e.cause) {
|
|
115
|
+
if (e instanceof UserError && e.code === 'ROBOTS_DISALLOWED')
|
|
116
|
+
return true;
|
|
117
|
+
}
|
|
118
|
+
return false;
|
|
119
|
+
}
|
|
112
120
|
export function siteName(value) {
|
|
113
121
|
if (!/^[a-zA-Z0-9][a-zA-Z0-9._-]*$/.test(value))
|
|
114
122
|
throw new UserError('INVALID_INPUT', 'site must be a name such as news.ycombinator.com, not a URL or path');
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { costSink } from '../measurement.js';
|
|
2
2
|
import { parseRobots, isPathAllowed } from './robots.js';
|
|
3
|
+
import { UserError } from '../local.js';
|
|
3
4
|
export const DEFAULT_USER_AGENT = 'webrecipe/0.1 (+https://github.com/Pillsoon/webrecipe)';
|
|
4
5
|
const LOOPBACK = /^(localhost|127\.0\.0\.1|\[::1\])(:\d+)?$/;
|
|
5
6
|
export class PolitenessLayer {
|
|
@@ -10,6 +11,7 @@ export class PolitenessLayer {
|
|
|
10
11
|
now;
|
|
11
12
|
sleep;
|
|
12
13
|
fetchImpl;
|
|
14
|
+
enforceRobots;
|
|
13
15
|
hosts = new Map();
|
|
14
16
|
constructor(opts = {}) {
|
|
15
17
|
this.userAgent = opts.userAgent ?? DEFAULT_USER_AGENT;
|
|
@@ -19,6 +21,7 @@ export class PolitenessLayer {
|
|
|
19
21
|
this.now = opts.now ?? (() => Date.now());
|
|
20
22
|
this.sleep = opts.sleep ?? ((ms) => new Promise((r) => setTimeout(r, ms)));
|
|
21
23
|
this.fetchImpl = opts.fetchImpl ?? fetch;
|
|
24
|
+
this.enforceRobots = opts.enforceRobots ?? false;
|
|
22
25
|
}
|
|
23
26
|
state(host) {
|
|
24
27
|
let s = this.hosts.get(host);
|
|
@@ -49,14 +52,20 @@ export class PolitenessLayer {
|
|
|
49
52
|
await this.robotsFor(u);
|
|
50
53
|
return this.state(u.host).intervalMs;
|
|
51
54
|
}
|
|
55
|
+
async refuseDisallowed(url) {
|
|
56
|
+
if (!(await this.isAllowed(url)))
|
|
57
|
+
throw new UserError('ROBOTS_DISALLOWED', `robots.txt disallows ${url}`);
|
|
58
|
+
}
|
|
52
59
|
/** Bypasses the rate limiter; used only to fetch robots.txt itself. */
|
|
53
|
-
async raw(url, opts = {}) {
|
|
60
|
+
async raw(url, opts = {}, check) {
|
|
54
61
|
const charge = costSink();
|
|
55
62
|
let current = url;
|
|
56
63
|
let method = opts.method ?? 'GET';
|
|
57
64
|
let requestBody = opts.body;
|
|
58
65
|
const headers = new Headers({ 'user-agent': this.userAgent, ...(opts.headers ?? {}) });
|
|
59
66
|
for (let redirects = 0;; redirects++) {
|
|
67
|
+
// Before every hop, not only the first: an allowed URL may redirect to a disallowed one.
|
|
68
|
+
await check?.(current);
|
|
60
69
|
charge({ networkRequests: 1 });
|
|
61
70
|
const res = await this.fetchImpl(current, {
|
|
62
71
|
method, headers: Object.fromEntries(headers.entries()), body: requestBody, redirect: 'manual', signal: AbortSignal.timeout(this.timeoutMs),
|
|
@@ -124,7 +133,7 @@ export class PolitenessLayer {
|
|
|
124
133
|
}
|
|
125
134
|
}
|
|
126
135
|
s.lastRequestAt = this.now();
|
|
127
|
-
const res = await this.raw(u.toString(), opts);
|
|
136
|
+
const res = await this.raw(u.toString(), opts, this.enforceRobots ? (url) => this.refuseDisallowed(url) : undefined);
|
|
128
137
|
if (res.status !== 429 || attempt >= this.maxRetries)
|
|
129
138
|
return { ...res, waitedMs };
|
|
130
139
|
const retryAfter = Number(res.headers['retry-after']);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { openSession } from '../browser/session.js';
|
|
2
2
|
import { BodyStore } from './body.js';
|
|
3
|
-
import { navigateAndSettle } from '../browser/navigate.js';
|
|
3
|
+
import { navigateAndSettle, guardPage } from '../browser/navigate.js';
|
|
4
4
|
/** A DOM mutation this soon after a response is treated as caused by it. */
|
|
5
5
|
const DOM_SETTLE_MS = 800;
|
|
6
6
|
/**
|
|
@@ -84,7 +84,7 @@ export function attributeMutations(requests, mutations) {
|
|
|
84
84
|
* A trace owns the files its large bodies were spilled into, so a caller that
|
|
85
85
|
* keeps the trace past the call must dispose of it.
|
|
86
86
|
*/
|
|
87
|
-
export async function record(plan, task, sites) {
|
|
87
|
+
export async function record(plan, task, sites, allowed) {
|
|
88
88
|
const origin = sites.origin(task.site);
|
|
89
89
|
const session = await openSession();
|
|
90
90
|
const actions = [];
|
|
@@ -124,7 +124,8 @@ export async function record(plan, task, sites) {
|
|
|
124
124
|
});
|
|
125
125
|
const target = plan.url(origin, task);
|
|
126
126
|
actions.push({ index: 0, type: 'navigate', value: target, at: Date.now() });
|
|
127
|
-
await
|
|
127
|
+
const guard = allowed ? await guardPage(session.page, allowed) : undefined;
|
|
128
|
+
await navigateAndSettle(session.page, target, plan.itemSelector, guard);
|
|
128
129
|
const mutations = (await session.page.evaluate('window.__fwaMutations || []'));
|
|
129
130
|
const completions = (await session.page.evaluate('window.__fwaCompletions || []'));
|
|
130
131
|
// Align first: attribution compares these timestamps against each other.
|
|
@@ -132,6 +133,7 @@ export async function record(plan, task, sites) {
|
|
|
132
133
|
attributeMutations(pending, mutations);
|
|
133
134
|
const finalHtml = await session.page.content();
|
|
134
135
|
await Promise.all(bodyReads);
|
|
136
|
+
guard?.check();
|
|
135
137
|
return {
|
|
136
138
|
site: task.site,
|
|
137
139
|
intent: task.intent,
|
package/dist/src/tasks.js
CHANGED
|
@@ -3,6 +3,8 @@ import { buildEngine } from './wiring.js';
|
|
|
3
3
|
import { loadLearnedPlans, mergePlans } from './authoring/plans.js';
|
|
4
4
|
import { PLANS } from '../benchmark/plans.js';
|
|
5
5
|
import { WILD_ORIGINS } from './sites.js';
|
|
6
|
+
import { buildUrl } from './executor/strategies/http-json.js';
|
|
7
|
+
import { measureResult } from './measurement.js';
|
|
6
8
|
/** The hand-written plans, with anything saved layered over them. */
|
|
7
9
|
export async function allPlans(planDir) {
|
|
8
10
|
const learned = await loadLearnedPlans(planDir);
|
|
@@ -21,11 +23,26 @@ export async function fetchTask(site, intent, input, opts = {}) {
|
|
|
21
23
|
const plan = plans[site]?.[intent];
|
|
22
24
|
if (!plan)
|
|
23
25
|
throw new UserError('NOT_TAUGHT', `No recipe saved as ${site}/${intent}. Run inspect on the page, then save.`);
|
|
26
|
+
const origin = origins[site] ?? WILD_ORIGINS[site];
|
|
27
|
+
const task = { id: opts.runId ?? 'fetch', site, intent, input };
|
|
24
28
|
// Reject missing placeholders before starting a browser or making a request.
|
|
25
|
-
plan.url(
|
|
26
|
-
const engine = buildEngine({ recipeDir: paths.recipes, plans, origins, heal: opts.heal });
|
|
29
|
+
const pageUrl = plan.url(origin, task);
|
|
30
|
+
const engine = buildEngine({ recipeDir: paths.recipes, plans, origins, heal: opts.heal, enforceRobots: true });
|
|
27
31
|
try {
|
|
28
|
-
|
|
32
|
+
// One measurement around both, so the robots.txt request is counted in meta
|
|
33
|
+
// whether the fetch goes ahead or is refused.
|
|
34
|
+
const outcome = await measureResult('browser', async () => {
|
|
35
|
+
// Decided here, before the executor, because the executor treats an HTTP
|
|
36
|
+
// failure as a reason to try the browser: a refusal raised any lower would
|
|
37
|
+
// be retried on the very page robots.txt excluded, and then relearned.
|
|
38
|
+
const recipe = await engine.registry.load(site, intent);
|
|
39
|
+
const urls = [pageUrl, ...(recipe && recipe.strategy.type !== 'browser' ? [buildUrl(origin, recipe, task)] : [])];
|
|
40
|
+
for (const url of urls) {
|
|
41
|
+
if (!(await engine.net.isAllowed(url)))
|
|
42
|
+
throw new UserError('ROBOTS_DISALLOWED', `robots.txt disallows ${url}`);
|
|
43
|
+
}
|
|
44
|
+
return engine.executor.run(task);
|
|
45
|
+
});
|
|
29
46
|
const verification = verifyReadable(outcome.items, Object.keys(plan.fields), verifications[site]?.[intent]);
|
|
30
47
|
// Derived from the checks, so a run that proves more says less.
|
|
31
48
|
const warnings = [...outcome.reasons, ...verificationWarnings(verification)];
|
package/dist/src/wiring.js
CHANGED
|
@@ -11,12 +11,14 @@ export function buildEngine(opts) {
|
|
|
11
11
|
const net = new PolitenessLayer({
|
|
12
12
|
userAgent: DEFAULT_USER_AGENT,
|
|
13
13
|
minIntervalMs: opts.minIntervalMs,
|
|
14
|
+
enforceRobots: opts.enforceRobots,
|
|
14
15
|
});
|
|
16
|
+
const guard = opts.enforceRobots ? (url) => net.isAllowed(url) : undefined;
|
|
15
17
|
const sites = new StaticSiteResolver({ ...WILD_ORIGINS, ...(opts.origins ?? {}) });
|
|
16
18
|
const registry = new RecipeRegistry(opts.recipeDir);
|
|
17
|
-
const browser = new BrowserStrategy(sites, opts.plans);
|
|
18
|
-
const warm = new WarmBrowserStrategy(sites, opts.plans);
|
|
19
|
-
const healer = new SelfHealer({ registry, sites, plans: opts.plans });
|
|
19
|
+
const browser = new BrowserStrategy(sites, opts.plans, true, guard);
|
|
20
|
+
const warm = new WarmBrowserStrategy(sites, opts.plans, guard);
|
|
21
|
+
const healer = new SelfHealer({ registry, sites, plans: opts.plans, guard });
|
|
20
22
|
const executor = new Executor({
|
|
21
23
|
registry,
|
|
22
24
|
strategies: [new HttpJsonStrategy(net, sites), new HttpHtmlStrategy(net, sites), warm, browser],
|