@crawlee/browser 4.0.0-beta.8 → 4.0.0-beta.81
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +0 -1
- package/index.js +0 -1
- package/internals/browser-crawler.d.ts +144 -86
- package/internals/browser-crawler.js +236 -230
- package/internals/browser-launcher.d.ts +11 -5
- package/internals/browser-launcher.js +11 -11
- package/package.json +7 -7
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/browser-crawler.d.ts.map +0 -1
- package/internals/browser-crawler.js.map +0 -1
- package/internals/browser-launcher.d.ts.map +0 -1
- package/internals/browser-launcher.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
|
@@ -1,8 +1,10 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { BrowserPool } from '@crawlee/browser-pool';
|
|
1
|
+
import { BasicCrawler, browserPoolCookieToToughCookie, ContextPipeline, cookieStringToToughCookie, enqueueLinks, handleRequestTimeout, NavigationSkippedError, RequestState, resolveBaseUrlForEnqueueLinksFiltering, SessionError, toughCookieToBrowserPoolCookie, tryAbsoluteURL, validators, } from '@crawlee/basic';
|
|
2
|
+
import { BrowserPool, RemoteBrowserPool } from '@crawlee/browser-pool';
|
|
3
3
|
import { CLOUDFLARE_RETRY_CSS_SELECTORS, RETRY_CSS_SELECTORS, sleep } from '@crawlee/utils';
|
|
4
4
|
import ow from 'ow';
|
|
5
|
-
import {
|
|
5
|
+
import { tryCancel } from '@apify/timeout';
|
|
6
|
+
const COOKIES_BEFORE_HOOKS = Symbol('cookiesBeforeHooks');
|
|
7
|
+
const readContextField = (ctx, key) => ctx[key];
|
|
6
8
|
/**
|
|
7
9
|
* Provides a simple framework for parallel crawling of web pages
|
|
8
10
|
* using headless browsers with [Puppeteer](https://github.com/puppeteer/puppeteer)
|
|
@@ -15,15 +17,18 @@ import { addTimeoutToPromise, tryCancel } from '@apify/timeout';
|
|
|
15
17
|
* If the target website doesn't need JavaScript, we should consider using the {@link CheerioCrawler},
|
|
16
18
|
* which downloads the pages using raw HTTP requests and is about 10x faster.
|
|
17
19
|
*
|
|
18
|
-
* The source URLs are represented by the {@link Request} objects that are fed from the
|
|
19
|
-
*
|
|
20
|
-
* constructor
|
|
20
|
+
* The source URLs are represented by the {@link Request} objects that are fed from the
|
|
21
|
+
* {@link IRequestManager|request manager} provided via the {@link BrowserCrawlerOptions.requestManager|`requestManager`}
|
|
22
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). If no `requestManager` is provided,
|
|
21
23
|
* the crawler will open the default request queue either when the {@link BrowserCrawler.addRequests|`crawler.addRequests()`} function is called,
|
|
22
24
|
* or if `requests` parameter (representing the initial requests) of the {@link BrowserCrawler.run|`crawler.run()`} function is provided.
|
|
23
25
|
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
26
|
+
* To read from a read-only source such as a {@link RequestList} while still being able to enqueue new requests,
|
|
27
|
+
* combine it with a queue into a {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`}
|
|
28
|
+
* and pass the result as `requestManager`.
|
|
29
|
+
*
|
|
30
|
+
* > The {@link BrowserCrawlerOptions.requestList|`requestList`} and {@link BrowserCrawlerOptions.requestQueue|`requestQueue`}
|
|
31
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
27
32
|
*
|
|
28
33
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
29
34
|
*
|
|
@@ -43,23 +48,24 @@ import { addTimeoutToPromise, tryCancel } from '@apify/timeout';
|
|
|
43
48
|
* @category Crawlers
|
|
44
49
|
*/
|
|
45
50
|
export class BrowserCrawler extends BasicCrawler {
|
|
46
|
-
config;
|
|
47
51
|
/**
|
|
48
|
-
* A reference to the underlying
|
|
49
|
-
*
|
|
52
|
+
* A reference to the underlying browser pool that manages the crawler's browsers. Typed as
|
|
53
|
+
* {@link IBrowserPool} so custom implementations can be plugged in via the `browserPool` constructor option.
|
|
50
54
|
*/
|
|
51
|
-
|
|
55
|
+
browserPool;
|
|
52
56
|
/**
|
|
53
|
-
*
|
|
57
|
+
* Set when the crawler constructed its own pool (a {@link BrowserPool}, or a {@link RemoteBrowserPool}
|
|
58
|
+
* built from the `remoteBrowser` option). Holds the same instance as `browserPool` but is the only reference
|
|
59
|
+
* the crawler tears down — a user-supplied `browserPool` is never owned and never destroyed by the crawler.
|
|
54
60
|
*/
|
|
55
|
-
|
|
61
|
+
ownedBrowserPool;
|
|
56
62
|
launchContext;
|
|
57
|
-
|
|
63
|
+
ignoreShadowRoots;
|
|
64
|
+
ignoreIframes;
|
|
58
65
|
navigationTimeoutMillis;
|
|
59
|
-
requestHandlerTimeoutInnerMillis;
|
|
60
66
|
preNavigationHooks;
|
|
61
67
|
postNavigationHooks;
|
|
62
|
-
|
|
68
|
+
saveResponseCookies;
|
|
63
69
|
static optionsShape = {
|
|
64
70
|
...BasicCrawler.optionsShape,
|
|
65
71
|
navigationTimeoutSecs: ow.optional.number.greaterThan(0),
|
|
@@ -67,65 +73,94 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
67
73
|
postNavigationHooks: ow.optional.array,
|
|
68
74
|
launchContext: ow.optional.object,
|
|
69
75
|
headless: ow.optional.any(ow.boolean, ow.string),
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
76
|
+
browserPool: ow.optional.object.validate(validators.browserPool),
|
|
77
|
+
remoteBrowser: ow.optional.object,
|
|
78
|
+
browserPoolOptions: ow.optional.object,
|
|
79
|
+
saveResponseCookies: ow.optional.boolean,
|
|
74
80
|
proxyConfiguration: ow.optional.object.validate(validators.proxyConfiguration),
|
|
75
|
-
ignoreShadowRoots: ow.optional.boolean,
|
|
76
|
-
ignoreIframes: ow.optional.boolean,
|
|
77
81
|
};
|
|
78
82
|
/**
|
|
79
83
|
* All `BrowserCrawler` parameters are passed via an options object.
|
|
80
84
|
*/
|
|
81
|
-
constructor(options
|
|
85
|
+
constructor(options) {
|
|
82
86
|
ow(options, 'BrowserCrawlerOptions', ow.object.exactShape(BrowserCrawler.optionsShape));
|
|
83
|
-
const { navigationTimeoutSecs = 60,
|
|
87
|
+
const { navigationTimeoutSecs = 60, saveResponseCookies = true, launchContext = {}, browserPool, remoteBrowser, browserPoolOptions, preNavigationHooks = [], postNavigationHooks = [], headless, ignoreIframes = false, ignoreShadowRoots = false, contextPipelineBuilder, extendContext, ...basicCrawlerOptions } = options;
|
|
88
|
+
const skipGuard = (action) => ({
|
|
89
|
+
action: async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {})),
|
|
90
|
+
});
|
|
84
91
|
super({
|
|
85
92
|
...basicCrawlerOptions,
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
93
|
+
contextPipelineBuilder: () => {
|
|
94
|
+
let pipeline = contextPipelineBuilder().compose({ action: this.prepareNavigation.bind(this) });
|
|
95
|
+
for (const hook of this.preNavigationHooks) {
|
|
96
|
+
pipeline = pipeline.compose(skipGuard(hook));
|
|
97
|
+
}
|
|
98
|
+
pipeline = pipeline.compose(skipGuard(this.navigate.bind(this)));
|
|
99
|
+
for (const hook of this.postNavigationHooks) {
|
|
100
|
+
pipeline = pipeline.compose(skipGuard(hook));
|
|
101
|
+
}
|
|
102
|
+
return pipeline
|
|
103
|
+
.compose(skipGuard(this.finalizeNavigation.bind(this)))
|
|
104
|
+
.compose({ action: this.handleBlockedRequestByContent.bind(this) })
|
|
105
|
+
.compose({ action: this.restoreRequestState.bind(this) });
|
|
106
|
+
},
|
|
107
|
+
extendContext: extendContext,
|
|
108
|
+
});
|
|
95
109
|
this.launchContext = launchContext;
|
|
96
110
|
this.navigationTimeoutMillis = navigationTimeoutSecs * 1000;
|
|
97
|
-
this.requestHandlerTimeoutInnerMillis = requestHandlerTimeoutSecs * 1000;
|
|
98
|
-
this.proxyConfiguration = proxyConfiguration;
|
|
99
111
|
this.preNavigationHooks = preNavigationHooks;
|
|
100
112
|
this.postNavigationHooks = postNavigationHooks;
|
|
113
|
+
this.ignoreIframes = ignoreIframes;
|
|
114
|
+
this.ignoreShadowRoots = ignoreShadowRoots;
|
|
101
115
|
if (headless != null) {
|
|
102
116
|
this.launchContext.launchOptions ??= {};
|
|
103
117
|
this.launchContext.launchOptions.headless = headless;
|
|
104
118
|
}
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
119
|
+
this.saveResponseCookies = saveResponseCookies;
|
|
120
|
+
// `browserPool` wins over `remoteBrowser` — a passed-in pool is used as-is, the sugar is ignored.
|
|
121
|
+
if (browserPool) {
|
|
122
|
+
this.browserPool = browserPool;
|
|
123
|
+
return;
|
|
110
124
|
}
|
|
125
|
+
const resolvedBrowserPoolOptions = browserPoolOptions ?? {};
|
|
111
126
|
if (launchContext?.userAgent) {
|
|
112
|
-
if (
|
|
127
|
+
if (resolvedBrowserPoolOptions.useFingerprints)
|
|
113
128
|
this.log.info('Custom user agent provided, disabling automatic browser fingerprint injection!');
|
|
114
|
-
|
|
129
|
+
resolvedBrowserPoolOptions.useFingerprints = false;
|
|
115
130
|
}
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
131
|
+
if (remoteBrowser) {
|
|
132
|
+
// The crawler already built the right plugin for its browser — hand it to a RemoteBrowserPool so the
|
|
133
|
+
// remote connection is always for the matching browser (no plugin to construct, no way to mismatch).
|
|
134
|
+
const { browserPlugins, ...remoteBrowserPoolOptions } = resolvedBrowserPoolOptions;
|
|
135
|
+
const remotePool = new RemoteBrowserPool({
|
|
136
|
+
browserPlugins: browserPlugins,
|
|
137
|
+
...remoteBrowser,
|
|
138
|
+
browserPoolOptions: remoteBrowserPoolOptions,
|
|
139
|
+
});
|
|
140
|
+
this.ownedBrowserPool = remotePool;
|
|
141
|
+
this.browserPool = remotePool;
|
|
142
|
+
return;
|
|
143
|
+
}
|
|
144
|
+
const ownedBrowserPool = new BrowserPool({
|
|
145
|
+
...resolvedBrowserPoolOptions,
|
|
121
146
|
});
|
|
147
|
+
this.ownedBrowserPool = ownedBrowserPool;
|
|
148
|
+
this.browserPool = ownedBrowserPool;
|
|
122
149
|
}
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
150
|
+
buildContextPipeline() {
|
|
151
|
+
return ContextPipeline.create().compose({
|
|
152
|
+
action: this.preparePage.bind(this),
|
|
153
|
+
cleanup: async (context) => {
|
|
154
|
+
context.registerDeferredCleanup(async () => {
|
|
155
|
+
const error = !context.session.isUsable()
|
|
156
|
+
? new SessionError('Session is no longer usable')
|
|
157
|
+
: undefined;
|
|
158
|
+
await this.browserPool
|
|
159
|
+
.closePage(context.page, { error })
|
|
160
|
+
.catch((closeError) => this.log.debug('Error while closing page', { error: closeError }));
|
|
161
|
+
});
|
|
162
|
+
},
|
|
163
|
+
});
|
|
129
164
|
}
|
|
130
165
|
async containsSelectors(page, selectors) {
|
|
131
166
|
const foundSelectors = (await Promise.all(selectors.map((selector) => page.$(selector))))
|
|
@@ -136,12 +171,6 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
136
171
|
}
|
|
137
172
|
async isRequestBlocked(crawlingContext) {
|
|
138
173
|
const { page, response } = crawlingContext;
|
|
139
|
-
const blockedStatusCodes =
|
|
140
|
-
// eslint-disable-next-line dot-notation
|
|
141
|
-
(this.sessionPool?.['blockedStatusCodes'].length ?? 0) > 0
|
|
142
|
-
? // eslint-disable-next-line dot-notation
|
|
143
|
-
this.sessionPool['blockedStatusCodes']
|
|
144
|
-
: DEFAULT_BLOCKED_STATUS_CODES;
|
|
145
174
|
// Cloudflare specific heuristic - wait 5 seconds if we get a 403 for the JS challenge to load / resolve.
|
|
146
175
|
if ((await this.containsSelectors(page, CLOUDFLARE_RETRY_CSS_SELECTORS)) && response?.status() === 403) {
|
|
147
176
|
await sleep(5000);
|
|
@@ -152,170 +181,173 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
152
181
|
return `Cloudflare challenge failed, found selectors: ${foundSelectors.join(', ')}`;
|
|
153
182
|
}
|
|
154
183
|
const foundSelectors = await this.containsSelectors(page, RETRY_CSS_SELECTORS);
|
|
155
|
-
const
|
|
184
|
+
const statusCode = response?.status() ?? 0;
|
|
156
185
|
if (foundSelectors)
|
|
157
186
|
return `Found selectors: ${foundSelectors.join(', ')}`;
|
|
158
|
-
if (
|
|
159
|
-
return `Received blocked status code: ${
|
|
187
|
+
if (this.blockedStatusCodes.has(statusCode))
|
|
188
|
+
return `Received blocked status code: ${statusCode}`;
|
|
160
189
|
return false;
|
|
161
190
|
}
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
*/
|
|
165
|
-
async _runRequestHandler(crawlingContext) {
|
|
166
|
-
const newPageOptions = {
|
|
191
|
+
async preparePage(crawlingContext) {
|
|
192
|
+
const page = await this.browserPool.newPage({
|
|
167
193
|
id: crawlingContext.id,
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
if (this.proxyConfiguration) {
|
|
171
|
-
const { session } = crawlingContext;
|
|
172
|
-
const proxyInfo = await this.proxyConfiguration.newProxyInfo(session?.id, {
|
|
173
|
-
request: crawlingContext.request,
|
|
174
|
-
});
|
|
175
|
-
crawlingContext.proxyInfo = proxyInfo;
|
|
176
|
-
newPageOptions.proxyUrl = proxyInfo?.url;
|
|
177
|
-
newPageOptions.proxyTier = proxyInfo?.proxyTier;
|
|
178
|
-
if (this.proxyConfiguration.isManInTheMiddle) {
|
|
179
|
-
/**
|
|
180
|
-
* @see https://playwright.dev/docs/api/class-browser/#browser-new-context
|
|
181
|
-
* @see https://github.com/puppeteer/puppeteer/blob/main/docs/api.md
|
|
182
|
-
*/
|
|
183
|
-
newPageOptions.pageOptions = {
|
|
184
|
-
ignoreHTTPSErrors: true,
|
|
185
|
-
acceptInsecureCerts: true,
|
|
186
|
-
};
|
|
187
|
-
}
|
|
188
|
-
}
|
|
189
|
-
const page = (await this.browserPool.newPage(newPageOptions));
|
|
190
|
-
tryCancel();
|
|
191
|
-
this._enhanceCrawlingContextWithPageInfo(crawlingContext, page, useIncognitoPages);
|
|
192
|
-
// DO NOT MOVE THIS LINE ABOVE!
|
|
193
|
-
// `enhanceCrawlingContextWithPageInfo` gives us a valid session.
|
|
194
|
-
// For example, `sessionPoolOptions.sessionOptions.maxUsageCount` can be `1`.
|
|
195
|
-
// So we must not save the session prior to making sure it was used only once, otherwise we would use it twice.
|
|
196
|
-
const { request, session } = crawlingContext;
|
|
197
|
-
if (!request.skipNavigation) {
|
|
198
|
-
await this._handleNavigation(crawlingContext);
|
|
199
|
-
tryCancel();
|
|
200
|
-
await this._responseHandler(crawlingContext);
|
|
201
|
-
tryCancel();
|
|
202
|
-
// save cookies
|
|
203
|
-
// TODO: Should we save the cookies also after/only the handle page?
|
|
204
|
-
if (this.persistCookiesPerSession) {
|
|
205
|
-
const cookies = await crawlingContext.browserController.getCookies(page);
|
|
206
|
-
tryCancel();
|
|
207
|
-
session?.setCookies(cookies, request.loadedUrl);
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
if (!this.requestMatchesEnqueueStrategy(request)) {
|
|
211
|
-
this.log.debug(
|
|
212
|
-
// eslint-disable-next-line dot-notation
|
|
213
|
-
`Skipping request ${request.id} (starting url: ${request.url} -> loaded url: ${request.loadedUrl}) because it does not match the enqueue strategy (${request['enqueueStrategy']}).`);
|
|
214
|
-
request.noRetry = true;
|
|
215
|
-
request.state = RequestState.SKIPPED;
|
|
216
|
-
return;
|
|
217
|
-
}
|
|
218
|
-
if (this.retryOnBlocked) {
|
|
219
|
-
const error = await this.isRequestBlocked(crawlingContext);
|
|
220
|
-
if (error)
|
|
221
|
-
throw new SessionError(error);
|
|
222
|
-
}
|
|
223
|
-
request.state = RequestState.REQUEST_HANDLER;
|
|
224
|
-
try {
|
|
225
|
-
await addTimeoutToPromise(async () => Promise.resolve(this.userProvidedRequestHandler(crawlingContext)), this.requestHandlerTimeoutInnerMillis, `requestHandler timed out after ${this.requestHandlerTimeoutInnerMillis / 1000} seconds.`);
|
|
226
|
-
request.state = RequestState.DONE;
|
|
227
|
-
}
|
|
228
|
-
catch (e) {
|
|
229
|
-
request.state = RequestState.ERROR;
|
|
230
|
-
throw e;
|
|
231
|
-
}
|
|
194
|
+
session: crawlingContext.session,
|
|
195
|
+
});
|
|
232
196
|
tryCancel();
|
|
197
|
+
const contextEnqueueLinks = crawlingContext.enqueueLinks;
|
|
198
|
+
return {
|
|
199
|
+
page,
|
|
200
|
+
get response() {
|
|
201
|
+
throw new Error("The `response` property is not available. This might mean that you're trying to access it before navigation or that navigation resulted in `null` (this should only happen with `about:` URLs)");
|
|
202
|
+
},
|
|
203
|
+
get gotoOptions() {
|
|
204
|
+
throw new Error('The `gotoOptions` property is not available until `prepareNavigation` runs.');
|
|
205
|
+
},
|
|
206
|
+
enqueueLinks: async (enqueueOptions = {}) => {
|
|
207
|
+
return (await browserCrawlerEnqueueLinks({
|
|
208
|
+
options: {
|
|
209
|
+
...enqueueOptions,
|
|
210
|
+
limit: await this.calculateEnqueuedRequestLimit(enqueueOptions?.limit),
|
|
211
|
+
},
|
|
212
|
+
page,
|
|
213
|
+
requestManager: await this.getRequestManager(),
|
|
214
|
+
robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
|
|
215
|
+
onSkippedRequest: this.handleSkippedRequest,
|
|
216
|
+
originalRequestUrl: crawlingContext.request.url,
|
|
217
|
+
finalRequestUrl: crawlingContext.request.loadedUrl,
|
|
218
|
+
enqueueLinks: contextEnqueueLinks,
|
|
219
|
+
})); // TODO make this type safe
|
|
220
|
+
},
|
|
221
|
+
};
|
|
233
222
|
}
|
|
234
|
-
|
|
235
|
-
crawlingContext.
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
223
|
+
async prepareNavigation(crawlingContext) {
|
|
224
|
+
if (crawlingContext.request.skipNavigation) {
|
|
225
|
+
return {
|
|
226
|
+
request: new Proxy(crawlingContext.request, {
|
|
227
|
+
get(target, propertyName, receiver) {
|
|
228
|
+
if (propertyName === 'loadedUrl') {
|
|
229
|
+
throw new NavigationSkippedError('The `request.loadedUrl` property is not available - `skipNavigation` was used');
|
|
230
|
+
}
|
|
231
|
+
return Reflect.get(target, propertyName, receiver);
|
|
232
|
+
},
|
|
233
|
+
}),
|
|
234
|
+
get response() {
|
|
235
|
+
throw new NavigationSkippedError('The `response` property is not available - `skipNavigation` was used');
|
|
236
|
+
},
|
|
237
|
+
};
|
|
247
238
|
}
|
|
248
|
-
crawlingContext.
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
requestQueue: await this.getRequestQueue(),
|
|
253
|
-
robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
|
|
254
|
-
onSkippedRequest: this.onSkippedRequest,
|
|
255
|
-
originalRequestUrl: crawlingContext.request.url,
|
|
256
|
-
finalRequestUrl: crawlingContext.request.loadedUrl,
|
|
257
|
-
});
|
|
239
|
+
crawlingContext.request.state = RequestState.BEFORE_NAV;
|
|
240
|
+
return {
|
|
241
|
+
gotoOptions: { timeout: this.navigationTimeoutMillis },
|
|
242
|
+
[COOKIES_BEFORE_HOOKS]: this._getCookieHeaderFromRequest(crawlingContext.request),
|
|
258
243
|
};
|
|
259
244
|
}
|
|
260
|
-
async
|
|
261
|
-
const gotoOptions = { timeout: this.navigationTimeoutMillis };
|
|
262
|
-
const preNavigationHooksCookies = this._getCookieHeaderFromRequest(crawlingContext.request);
|
|
263
|
-
crawlingContext.request.state = RequestState.BEFORE_NAV;
|
|
264
|
-
await this._executeHooks(this.preNavigationHooks, crawlingContext, gotoOptions);
|
|
245
|
+
async navigate(crawlingContext) {
|
|
265
246
|
tryCancel();
|
|
266
|
-
const
|
|
267
|
-
|
|
247
|
+
const gotoOptions = crawlingContext.gotoOptions;
|
|
248
|
+
const cookiesBeforeHooks = readContextField(crawlingContext, COOKIES_BEFORE_HOOKS);
|
|
249
|
+
const cookiesAfterHooks = this._getCookieHeaderFromRequest(crawlingContext.request);
|
|
250
|
+
await this.applyCookies(crawlingContext, cookiesBeforeHooks, cookiesAfterHooks);
|
|
251
|
+
let response;
|
|
268
252
|
try {
|
|
269
|
-
|
|
253
|
+
response = (await this._navigationHandler(crawlingContext, gotoOptions)) ?? undefined;
|
|
270
254
|
}
|
|
271
255
|
catch (error) {
|
|
272
|
-
await this.
|
|
256
|
+
await this.handleNavigationTimeout(crawlingContext, error);
|
|
273
257
|
crawlingContext.request.state = RequestState.ERROR;
|
|
274
|
-
this.
|
|
258
|
+
this.throwIfProxyError(error);
|
|
275
259
|
throw error;
|
|
276
260
|
}
|
|
277
261
|
tryCancel();
|
|
278
262
|
crawlingContext.request.state = RequestState.AFTER_NAV;
|
|
279
|
-
|
|
263
|
+
return { response };
|
|
264
|
+
}
|
|
265
|
+
async finalizeNavigation(crawlingContext) {
|
|
266
|
+
tryCancel();
|
|
267
|
+
let response;
|
|
268
|
+
try {
|
|
269
|
+
response = crawlingContext.response;
|
|
270
|
+
}
|
|
271
|
+
catch {
|
|
272
|
+
// `preparePage` installs a throwing getter for `response`; reaching this branch means
|
|
273
|
+
// navigation produced no response and no hook overrode it. Treat as undefined.
|
|
274
|
+
}
|
|
275
|
+
await this.processResponse(response, crawlingContext);
|
|
276
|
+
tryCancel();
|
|
277
|
+
// TODO: Should we save the cookies also after/only the handle page?
|
|
278
|
+
if (this.saveResponseCookies && crawlingContext.session) {
|
|
279
|
+
const { cookies } = await this.browserPool.extractPageState(crawlingContext.page);
|
|
280
|
+
tryCancel();
|
|
281
|
+
const url = crawlingContext.request.loadedUrl;
|
|
282
|
+
for (const cookie of cookies) {
|
|
283
|
+
try {
|
|
284
|
+
crawlingContext.session.cookieJar.setCookieSync(browserPoolCookieToToughCookie(cookie), url, {
|
|
285
|
+
ignoreError: false,
|
|
286
|
+
});
|
|
287
|
+
}
|
|
288
|
+
catch (e) {
|
|
289
|
+
this.log.debug(`Could not set cookie: ${e.message}`);
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
return { request: crawlingContext.request };
|
|
294
|
+
}
|
|
295
|
+
async handleBlockedRequestByContent(crawlingContext) {
|
|
296
|
+
if (this.retryOnBlocked) {
|
|
297
|
+
const error = await this.isRequestBlocked(crawlingContext);
|
|
298
|
+
if (error)
|
|
299
|
+
throw new SessionError(error);
|
|
300
|
+
}
|
|
301
|
+
return {};
|
|
280
302
|
}
|
|
281
|
-
async
|
|
282
|
-
|
|
303
|
+
async restoreRequestState(crawlingContext) {
|
|
304
|
+
crawlingContext.request.state = RequestState.REQUEST_HANDLER;
|
|
305
|
+
return {};
|
|
306
|
+
}
|
|
307
|
+
async applyCookies({ session, request, page }, preHooksCookies, postHooksCookies) {
|
|
308
|
+
const sessionCookie = session?.cookieJar.getCookiesSync(request.url).map(toughCookieToBrowserPoolCookie) ?? [];
|
|
283
309
|
const parsedPreHooksCookies = preHooksCookies.split(/ *; */).map((c) => cookieStringToToughCookie(c));
|
|
284
310
|
const parsedPostHooksCookies = postHooksCookies.split(/ *; */).map((c) => cookieStringToToughCookie(c));
|
|
285
|
-
|
|
311
|
+
const cookies = [...sessionCookie, ...parsedPreHooksCookies, ...parsedPostHooksCookies]
|
|
286
312
|
.filter((c) => typeof c !== 'undefined' && c !== null)
|
|
287
|
-
.map((c) => ({ ...c, url: c.domain ? undefined : request.url }))
|
|
313
|
+
.map((c) => ({ ...c, url: c.domain ? undefined : request.url }));
|
|
314
|
+
await this.browserPool.injectPageState(page, { cookies });
|
|
288
315
|
}
|
|
289
316
|
/**
|
|
290
|
-
* Marks session bad in
|
|
317
|
+
* Marks session bad on navigation timeout, and stops in-flight page loading on any navigation error.
|
|
291
318
|
*/
|
|
292
|
-
async
|
|
293
|
-
const { session } = crawlingContext;
|
|
294
|
-
if (error
|
|
319
|
+
async handleNavigationTimeout(crawlingContext, error) {
|
|
320
|
+
const { session, page } = crawlingContext;
|
|
321
|
+
if (error?.constructor.name === 'TimeoutError') {
|
|
295
322
|
handleRequestTimeout({ session, errorMessage: error.message });
|
|
296
323
|
}
|
|
297
|
-
|
|
324
|
+
// Fire-and-forget: no user code will run on this page after a failed navigation.
|
|
325
|
+
// Swallow rejections: the page may already be detached.
|
|
326
|
+
void page.evaluate(() => window.stop()).catch(() => { });
|
|
298
327
|
}
|
|
299
328
|
/**
|
|
300
329
|
* Transforms proxy-related errors to `SessionError`.
|
|
301
330
|
*/
|
|
302
|
-
|
|
331
|
+
throwIfProxyError(error) {
|
|
303
332
|
if (this.isProxyError(error)) {
|
|
304
333
|
throw new SessionError(this._getMessageFromError(error));
|
|
305
334
|
}
|
|
306
335
|
}
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
*/
|
|
310
|
-
async _responseHandler(crawlingContext) {
|
|
311
|
-
const { response, session, request, page } = crawlingContext;
|
|
336
|
+
async processResponse(response, crawlingContext) {
|
|
337
|
+
const { session, request, page } = crawlingContext;
|
|
312
338
|
if (typeof response === 'object' && typeof response.status === 'function') {
|
|
313
339
|
const status = response.status();
|
|
314
340
|
this.stats.registerStatusCode(status);
|
|
341
|
+
if (this.isErrorStatusCode(status)) {
|
|
342
|
+
if (this.additionalHttpErrorStatusCodes.has(status)) {
|
|
343
|
+
throw new Error(`${status} - Error status code was set by user.`);
|
|
344
|
+
}
|
|
345
|
+
throw new Error(`${status} - Internal Server Error`);
|
|
346
|
+
}
|
|
315
347
|
}
|
|
316
348
|
if (this.sessionPool && response && session) {
|
|
317
349
|
if (typeof response === 'object' && typeof response.status === 'function') {
|
|
318
|
-
this._throwOnBlockedRequest(
|
|
350
|
+
this._throwOnBlockedRequest(response.status());
|
|
319
351
|
}
|
|
320
352
|
else {
|
|
321
353
|
this.log.debug('Got a malformed Browser response.', { request, response });
|
|
@@ -323,68 +355,43 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
323
355
|
}
|
|
324
356
|
request.loadedUrl = await page.url();
|
|
325
357
|
}
|
|
326
|
-
async _extendLaunchContext(_pageId, launchContext) {
|
|
327
|
-
const launchContextExtends = {};
|
|
328
|
-
if (this.sessionPool) {
|
|
329
|
-
launchContextExtends.session = await this.sessionPool.getSession();
|
|
330
|
-
}
|
|
331
|
-
if (this.proxyConfiguration && !launchContext.proxyUrl) {
|
|
332
|
-
const proxyInfo = await this.proxyConfiguration.newProxyInfo(launchContextExtends.session?.id, {
|
|
333
|
-
proxyTier: launchContext.proxyTier ?? undefined,
|
|
334
|
-
});
|
|
335
|
-
launchContext.proxyUrl = proxyInfo?.url;
|
|
336
|
-
launchContextExtends.proxyInfo = proxyInfo;
|
|
337
|
-
// Disable SSL verification for MITM proxies
|
|
338
|
-
if (this.proxyConfiguration.isManInTheMiddle) {
|
|
339
|
-
/**
|
|
340
|
-
* @see https://playwright.dev/docs/api/class-browser/#browser-new-context
|
|
341
|
-
* @see https://github.com/puppeteer/puppeteer/blob/main/docs/api.md
|
|
342
|
-
*/
|
|
343
|
-
launchContext.launchOptions.ignoreHTTPSErrors = true;
|
|
344
|
-
launchContext.launchOptions.acceptInsecureCerts = true;
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
launchContext.extend(launchContextExtends);
|
|
348
|
-
}
|
|
349
|
-
_maybeAddSessionRetiredListener(_pageId, browserController) {
|
|
350
|
-
if (this.sessionPool) {
|
|
351
|
-
const listener = (session) => {
|
|
352
|
-
const { launchContext } = browserController;
|
|
353
|
-
if (session.id === launchContext.session.id) {
|
|
354
|
-
this.browserPool.retireBrowserController(browserController);
|
|
355
|
-
}
|
|
356
|
-
};
|
|
357
|
-
this.sessionPool.on(EVENT_SESSION_RETIRED, listener);
|
|
358
|
-
browserController.on("browserClosed" /* BROWSER_CONTROLLER_EVENTS.BROWSER_CLOSED */, () => {
|
|
359
|
-
return this.sessionPool.removeListener(EVENT_SESSION_RETIRED, listener);
|
|
360
|
-
});
|
|
361
|
-
}
|
|
362
|
-
}
|
|
363
358
|
/**
|
|
364
359
|
* Function for cleaning up after all requests are processed.
|
|
365
360
|
* @ignore
|
|
366
361
|
*/
|
|
367
362
|
async teardown() {
|
|
368
|
-
await this.
|
|
363
|
+
await this.ownedBrowserPool?.destroy();
|
|
369
364
|
await super.teardown();
|
|
370
365
|
}
|
|
371
366
|
}
|
|
372
367
|
/** @internal */
|
|
373
|
-
|
|
368
|
+
function containsEnqueueLinks(options) {
|
|
369
|
+
return !!options.enqueueLinks;
|
|
370
|
+
}
|
|
371
|
+
/** @internal */
|
|
372
|
+
export async function browserCrawlerEnqueueLinks(options) {
|
|
373
|
+
const { options: enqueueLinksOptions, finalRequestUrl, originalRequestUrl, page } = options;
|
|
374
374
|
const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
|
|
375
|
-
enqueueStrategy:
|
|
375
|
+
enqueueStrategy: enqueueLinksOptions?.strategy,
|
|
376
376
|
finalRequestUrl,
|
|
377
377
|
originalRequestUrl,
|
|
378
|
-
userProvidedBaseUrl:
|
|
378
|
+
userProvidedBaseUrl: enqueueLinksOptions?.baseUrl,
|
|
379
379
|
});
|
|
380
|
-
const urls = await extractUrlsFromPage(page,
|
|
380
|
+
const urls = await extractUrlsFromPage(page, enqueueLinksOptions?.selector ?? 'a', enqueueLinksOptions?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
|
|
381
|
+
if (containsEnqueueLinks(options)) {
|
|
382
|
+
return options.enqueueLinks({
|
|
383
|
+
urls,
|
|
384
|
+
baseUrl,
|
|
385
|
+
...enqueueLinksOptions,
|
|
386
|
+
});
|
|
387
|
+
}
|
|
381
388
|
return enqueueLinks({
|
|
382
|
-
|
|
383
|
-
robotsTxtFile,
|
|
384
|
-
onSkippedRequest,
|
|
389
|
+
requestManager: options.requestManager,
|
|
390
|
+
robotsTxtFile: options.robotsTxtFile,
|
|
391
|
+
onSkippedRequest: options.onSkippedRequest,
|
|
385
392
|
urls,
|
|
386
393
|
baseUrl,
|
|
387
|
-
...
|
|
394
|
+
...enqueueLinksOptions,
|
|
388
395
|
});
|
|
389
396
|
}
|
|
390
397
|
/**
|
|
@@ -412,4 +419,3 @@ page, selector, baseUrl) {
|
|
|
412
419
|
})
|
|
413
420
|
.filter((href) => !!href);
|
|
414
421
|
}
|
|
415
|
-
//# sourceMappingURL=browser-crawler.js.map
|
|
@@ -44,6 +44,11 @@ export interface BrowserLaunchContext<TOptions, Launcher> extends BrowserPluginO
|
|
|
44
44
|
* to reduce the chance of detection of the crawler.
|
|
45
45
|
*/
|
|
46
46
|
userAgent?: string;
|
|
47
|
+
/**
|
|
48
|
+
* If set to `true`, TLS certificate errors from the upstream proxy will be ignored.
|
|
49
|
+
* This is useful when using HTTPS proxies with self-signed certificates.
|
|
50
|
+
*/
|
|
51
|
+
ignoreProxyCertificate?: boolean;
|
|
47
52
|
/**
|
|
48
53
|
* The type of browser to be launched.
|
|
49
54
|
* By default, `chromium` is used. Other browsers like `webkit` or `firefox` can be used.
|
|
@@ -81,6 +86,8 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
81
86
|
useIncognitoPages: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
82
87
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
83
88
|
browserPerProxy: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
89
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
90
|
+
ignoreProxyCertificate: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
84
91
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
85
92
|
userDataDir: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
|
|
86
93
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
@@ -103,12 +110,11 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
103
110
|
*/
|
|
104
111
|
launch(): LaunchResult;
|
|
105
112
|
createLaunchOptions(): Dictionary;
|
|
106
|
-
protected
|
|
107
|
-
|
|
113
|
+
protected getDefaultHeadlessOption(): boolean;
|
|
114
|
+
private getChromeExecutablePath;
|
|
108
115
|
/**
|
|
109
116
|
* Gets a typical path to Chrome executable, depending on the current operating system.
|
|
110
117
|
*/
|
|
111
|
-
|
|
112
|
-
|
|
118
|
+
private getTypicalChromeExecutablePath;
|
|
119
|
+
private validateProxyUrlProtocol;
|
|
113
120
|
}
|
|
114
|
-
//# sourceMappingURL=browser-launcher.d.ts.map
|