@crawlee/browser 4.0.0-rc.0 → 4.0.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/internals/browser-crawler.d.ts +39 -59
- package/internals/browser-crawler.js +73 -67
- package/internals/browser-launcher.d.ts +8 -13
- package/internals/browser-launcher.js +15 -10
- package/package.json +9 -10
package/README.md
CHANGED
|
@@ -34,7 +34,7 @@ Crawlee is available as the [`crawlee`](https://www.npmjs.com/package/crawlee) N
|
|
|
34
34
|
|
|
35
35
|
We recommend visiting the [Introduction tutorial](https://crawlee.dev/js/docs/introduction) in Crawlee documentation for more information.
|
|
36
36
|
|
|
37
|
-
> Crawlee requires **Node.js
|
|
37
|
+
> Crawlee requires **Node.js 22.13 or higher**.
|
|
38
38
|
|
|
39
39
|
### With Crawlee CLI
|
|
40
40
|
|
|
@@ -1,9 +1,8 @@
|
|
|
1
|
-
import type { AddRequestsBatchedResult, BasicCrawlerOptions, CrawlingContext, EnqueueLinksOptions, ErrorHandler, ExtractLinksOptions, GetUserDataFromRequest, LoadedRequest,
|
|
1
|
+
import type { AddRequestsBatchedResult, BasicCrawlerOptions, CrawlingContext, EnqueueLinksOptions, ErrorHandler, ExtractLinksOptions, GetUserDataFromRequest, LoadedRequest, CrawlingRequest, RequestHandler, RouterHandler } from '@crawlee/basic';
|
|
2
2
|
import { BasicCrawler, ContextPipeline } from '@crawlee/basic';
|
|
3
3
|
import type { CommonPage, CrawlerRemoteBrowserOptions } from '@crawlee/browser-pool';
|
|
4
4
|
import type { Awaitable, Dictionary, IBrowserPool } from '@crawlee/types';
|
|
5
5
|
import { z } from 'zod';
|
|
6
|
-
import type { BrowserLaunchContext } from './browser-launcher.js';
|
|
7
6
|
interface BaseResponse {
|
|
8
7
|
status(): number;
|
|
9
8
|
/** Optional because only Playwright and Puppeteer responses are guaranteed to carry it. */
|
|
@@ -11,18 +10,14 @@ interface BaseResponse {
|
|
|
11
10
|
}
|
|
12
11
|
/**
|
|
13
12
|
* The type of a browser pool the crawler builds (and therefore owns) for itself. It's an {@link IBrowserPool} that
|
|
14
|
-
* additionally exposes
|
|
15
|
-
* itself intentionally omits `
|
|
13
|
+
* additionally exposes the lifecycle hooks a crawler only ever calls on a pool it created — which is why
|
|
14
|
+
* {@link IBrowserPool} itself intentionally omits them: `releaseAllBrowsers()` at the end of every run, and
|
|
15
|
+
* `destroy()` once the crawler itself is destroyed.
|
|
16
16
|
*/
|
|
17
17
|
export type OwnedBrowserPool<Page> = IBrowserPool<Page> & {
|
|
18
|
+
releaseAllBrowsers: () => Promise<void>;
|
|
18
19
|
destroy: () => Promise<void>;
|
|
19
20
|
};
|
|
20
|
-
/**
|
|
21
|
-
* Rejects options that exist only to configure the browser pool the crawler would have built for itself.
|
|
22
|
-
* Accepting them alongside a pre-built `browserPool` and quietly ignoring them is how `browserPoolOptions` grew
|
|
23
|
-
* into a second, half-working way of configuring the same pool.
|
|
24
|
-
*/
|
|
25
|
-
export declare function assertBrowserPoolNotConfigured(crawlerName: string, ignoredOptions: Dictionary): void;
|
|
26
21
|
export interface BrowserCrawlingContext<Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
27
22
|
GoToOptions extends Dictionary = Dictionary> extends CrawlingContext<UserData> {
|
|
28
23
|
/**
|
|
@@ -32,7 +27,7 @@ GoToOptions extends Dictionary = Dictionary> extends CrawlingContext<UserData> {
|
|
|
32
27
|
/**
|
|
33
28
|
* The request object that was successfully loaded and navigated to, including the {@link Request.loadedUrl|`loadedUrl`} property.
|
|
34
29
|
*/
|
|
35
|
-
request: LoadedRequest<
|
|
30
|
+
request: LoadedRequest<CrawlingRequest<UserData>>;
|
|
36
31
|
/**
|
|
37
32
|
* The HTTP response object returned by the browser's navigation.
|
|
38
33
|
*/
|
|
@@ -52,8 +47,7 @@ GoToOptions extends Dictionary = Dictionary> extends CrawlingContext<UserData> {
|
|
|
52
47
|
enqueueLinks: (options?: EnqueueLinksOptions) => Promise<AddRequestsBatchedResult>;
|
|
53
48
|
}
|
|
54
49
|
export type BrowserHook<Context = BrowserCrawlingContext, ContextExtension = {}> = (crawlingContext: Context & ContextExtension) => Awaitable<void | Partial<Context>>;
|
|
55
|
-
export interface BrowserCrawlerOptions<Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse, Context extends BrowserCrawlingContext<Page, Response> = BrowserCrawlingContext<Page, Response>, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends Omit<BasicCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'requestHandler' | 'failedRequestHandler' | 'errorHandler'> {
|
|
56
|
-
launchContext?: BrowserLaunchContext<any, any>;
|
|
50
|
+
export interface BrowserCrawlerOptions<Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse, Context extends BrowserCrawlingContext<Page, Response> = BrowserCrawlingContext<Page, Response>, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends Omit<BasicCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'requestHandler' | 'failedRequestHandler' | 'errorHandler' | 'contextPipelineBuilder'> {
|
|
57
51
|
/**
|
|
58
52
|
* The browser pool the crawler should serve its pages from. This is the single way to run a pool with
|
|
59
53
|
* non-default options: build one with the factory that matches your crawler
|
|
@@ -105,7 +99,7 @@ export interface BrowserCrawlerOptions<Page extends CommonPage = CommonPage, Res
|
|
|
105
99
|
* To make this work, we should **always**
|
|
106
100
|
* let our function throw exceptions rather than catch them.
|
|
107
101
|
* The exceptions are logged to the request using the
|
|
108
|
-
* {@link
|
|
102
|
+
* {@link CrawlingRequest.pushErrorMessage|`request.pushErrorMessage()`} function.
|
|
109
103
|
*/
|
|
110
104
|
requestHandler?: RouterHandler<ExtendedContext, Routes> | RequestHandler<ExtendedContext>;
|
|
111
105
|
/**
|
|
@@ -244,28 +238,19 @@ export interface BrowserCrawlerOptions<Page extends CommonPage = CommonPage, Res
|
|
|
244
238
|
*
|
|
245
239
|
* @category Crawlers
|
|
246
240
|
*/
|
|
247
|
-
export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse,
|
|
241
|
+
export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPage, Response extends BaseResponse = BaseResponse, Context extends BrowserCrawlingContext<Page, Response> = BrowserCrawlingContext<Page, Response>, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}, GoToOptions extends Dictionary = Dictionary> extends BasicCrawler<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
248
242
|
#private;
|
|
249
243
|
/**
|
|
250
244
|
* A reference to the underlying browser pool that manages the crawler's browsers. Typed as
|
|
251
245
|
* {@link IBrowserPool} so custom implementations can be plugged in via the `browserPool` constructor option.
|
|
252
246
|
*/
|
|
253
247
|
get browserPool(): IBrowserPool<Page>;
|
|
254
|
-
launchContext: BrowserLaunchContext<LaunchOptions, unknown>;
|
|
255
248
|
protected readonly ignoreShadowRoots: boolean;
|
|
256
249
|
protected readonly ignoreIframes: boolean;
|
|
250
|
+
/**
|
|
251
|
+
* @internal
|
|
252
|
+
*/
|
|
257
253
|
protected static optionsShape: {
|
|
258
|
-
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
259
|
-
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
260
|
-
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
261
|
-
launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
|
|
262
|
-
browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
263
|
-
browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
264
|
-
remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
265
|
-
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
266
|
-
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
267
|
-
ignoreIframes: z.ZodDefault<z.ZodBoolean>;
|
|
268
|
-
ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
|
|
269
254
|
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
270
255
|
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
271
256
|
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
@@ -297,25 +282,23 @@ export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPag
|
|
|
297
282
|
}, z.core.$strict>]>>;
|
|
298
283
|
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
299
284
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
300
|
-
httpClient: z.ZodOptional<z.
|
|
285
|
+
httpClient: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/http-client").BaseHttpClient>>;
|
|
301
286
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
302
|
-
configuration: z.ZodOptional<z.
|
|
287
|
+
configuration: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/basic").Configuration>>;
|
|
303
288
|
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
304
289
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
305
|
-
eventManager: z.ZodOptional<z.
|
|
290
|
+
eventManager: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/basic").EventManager>>;
|
|
306
291
|
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
307
292
|
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
308
293
|
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
294
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
309
295
|
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
310
296
|
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
311
297
|
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
312
298
|
id: z.ZodOptional<z.ZodString>;
|
|
313
|
-
};
|
|
314
|
-
protected static optionsSchema: z.ZodObject<{
|
|
315
299
|
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
316
300
|
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
317
301
|
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
318
|
-
launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
|
|
319
302
|
browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
320
303
|
browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
321
304
|
remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
@@ -323,6 +306,9 @@ export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPag
|
|
|
323
306
|
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
324
307
|
ignoreIframes: z.ZodDefault<z.ZodBoolean>;
|
|
325
308
|
ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
|
|
309
|
+
};
|
|
310
|
+
/** @internal */
|
|
311
|
+
protected static optionsSchema: z.ZodObject<{
|
|
326
312
|
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
327
313
|
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
328
314
|
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
@@ -354,22 +340,34 @@ export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPag
|
|
|
354
340
|
}, z.core.$strict>]>>;
|
|
355
341
|
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
356
342
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
357
|
-
httpClient: z.ZodOptional<z.
|
|
343
|
+
httpClient: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/http-client").BaseHttpClient>>;
|
|
358
344
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
359
|
-
configuration: z.ZodOptional<z.
|
|
345
|
+
configuration: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/basic").Configuration>>;
|
|
360
346
|
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
361
347
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
362
|
-
eventManager: z.ZodOptional<z.
|
|
348
|
+
eventManager: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/basic").EventManager>>;
|
|
363
349
|
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
364
350
|
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
365
351
|
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
352
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
366
353
|
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
367
354
|
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
368
355
|
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
369
356
|
id: z.ZodOptional<z.ZodString>;
|
|
357
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
358
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
359
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
360
|
+
browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
361
|
+
browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
362
|
+
remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
363
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
364
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
365
|
+
ignoreIframes: z.ZodDefault<z.ZodBoolean>;
|
|
366
|
+
ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
|
|
370
367
|
}, z.core.$strict>;
|
|
371
368
|
/**
|
|
372
369
|
* All `BrowserCrawler` parameters are passed via an options object.
|
|
370
|
+
* @internal
|
|
373
371
|
*/
|
|
374
372
|
protected constructor(options: BrowserCrawlerOptions<Page, Response, Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> & {
|
|
375
373
|
contextPipelineBuilder: () => ContextPipeline<CrawlingContext, Context>;
|
|
@@ -380,41 +378,23 @@ export declare abstract class BrowserCrawler<Page extends CommonPage = CommonPag
|
|
|
380
378
|
*/
|
|
381
379
|
browserPoolBuilder: (remoteBrowser?: CrawlerRemoteBrowserOptions) => OwnedBrowserPool<Page>;
|
|
382
380
|
});
|
|
381
|
+
/** @internal */
|
|
383
382
|
protected getNavigationTimeoutMillis(): number;
|
|
384
383
|
protected buildContextPipeline(): ContextPipeline<CrawlingContext, BrowserCrawlingContext<Page, Response, Dictionary>>;
|
|
385
|
-
private containsSelectors;
|
|
386
|
-
private isRequestBlocked;
|
|
387
|
-
private preparePage;
|
|
388
|
-
private prepareNavigation;
|
|
389
384
|
private navigate;
|
|
390
|
-
private finalizeNavigation;
|
|
391
|
-
/**
|
|
392
|
-
* Copies cookies from the live browser page into the session cookie jar.
|
|
393
|
-
*/
|
|
394
|
-
private persistCookiesFromPage;
|
|
395
385
|
/**
|
|
396
386
|
* Runs the user request handler, then re-reads browser cookies so login flows /
|
|
397
387
|
* `page.setCookie` / XHR `Set-Cookie` updates are stored for later requests.
|
|
388
|
+
* @internal
|
|
398
389
|
*/
|
|
399
390
|
protected runRequestHandler(crawlingContext: ExtendedContext): Promise<void>;
|
|
400
|
-
private handleBlockedRequestByContent;
|
|
401
|
-
private restoreRequestState;
|
|
402
|
-
private applyCookies;
|
|
403
|
-
/**
|
|
404
|
-
* Marks session bad on navigation timeout, and stops in-flight page loading on any navigation error.
|
|
405
|
-
*/
|
|
406
|
-
private handleNavigationTimeout;
|
|
407
|
-
/**
|
|
408
|
-
* Transforms proxy-related errors to `SessionError`.
|
|
409
|
-
*/
|
|
410
|
-
private throwIfProxyError;
|
|
411
391
|
protected abstract navigationHandler(crawlingContext: BrowserCrawlingContext<Page, Response>, gotoOptions: GoToOptions): Promise<Context['response'] | null | undefined>;
|
|
412
|
-
private processResponse;
|
|
413
392
|
/**
|
|
414
|
-
*
|
|
415
|
-
*
|
|
393
|
+
* Closes the browsers of a pool the crawler owns, so a finished run leaves none behind. The pool itself is
|
|
394
|
+
* crawler-lifetime and survives — destroying it here would hand a repeated `run()` a dead pool.
|
|
416
395
|
*/
|
|
417
396
|
teardown(): Promise<void>;
|
|
397
|
+
destroy(): Promise<void>;
|
|
418
398
|
}
|
|
419
399
|
/**
|
|
420
400
|
* Extracts URLs from a given page.
|
|
@@ -1,22 +1,8 @@
|
|
|
1
|
-
import { BasicCrawler, browserPoolCookieToToughCookie, ContextPipeline, cookieStringToToughCookie, EnqueueStrategy, NavigationSkippedError, OwnedOrInjected,
|
|
2
|
-
import { CLOUDFLARE_RETRY_CSS_SELECTORS, RETRY_CSS_SELECTORS } from '@crawlee/utils/internal';
|
|
1
|
+
import { BasicCrawler, browserPoolCookieToToughCookie, ContextPipeline, cookieStringToToughCookie, EnqueueStrategy, NavigationSkippedError, OwnedOrInjected, remainingNavigationWindowMillis, RequestState, ContextPipelineInitializationError, RequestHandlerError, RequestThrottledError, resolveBaseUrlForEnqueueLinksFiltering, SessionError, SessionRetiredError, toughCookieToBrowserPoolCookie, validators, } from '@crawlee/basic';
|
|
2
|
+
import { assertBrowserPoolNotConfigured, CLOUDFLARE_RETRY_CSS_SELECTORS, parseArgument, RETRY_CSS_SELECTORS, schemas, tryAbsoluteURL, } from '@crawlee/utils/internal';
|
|
3
3
|
import { sleep } from '@crawlee/utils';
|
|
4
4
|
import { z } from 'zod';
|
|
5
5
|
import { addTimeoutToPromise, TimeoutError, tryCancel } from '@apify/timeout';
|
|
6
|
-
/**
|
|
7
|
-
* Rejects options that exist only to configure the browser pool the crawler would have built for itself.
|
|
8
|
-
* Accepting them alongside a pre-built `browserPool` and quietly ignoring them is how `browserPoolOptions` grew
|
|
9
|
-
* into a second, half-working way of configuring the same pool.
|
|
10
|
-
*/
|
|
11
|
-
export function assertBrowserPoolNotConfigured(crawlerName, ignoredOptions) {
|
|
12
|
-
const names = Object.keys(ignoredOptions).filter((name) => ignoredOptions[name] !== undefined);
|
|
13
|
-
if (names.length === 0) {
|
|
14
|
-
return;
|
|
15
|
-
}
|
|
16
|
-
throw new Error(`${crawlerName}: ${names.map((name) => `\`${name}\``).join(', ')} cannot be combined with \`browserPool\`, ` +
|
|
17
|
-
`${names.length > 1 ? 'they configure' : 'it configures'} the browser pool the crawler would build for ` +
|
|
18
|
-
'itself. Configure the pool you pass in instead.');
|
|
19
|
-
}
|
|
20
6
|
const COOKIES_BEFORE_HOOKS = Symbol('cookiesBeforeHooks');
|
|
21
7
|
const readContextField = (ctx, key) => ctx[key];
|
|
22
8
|
/**
|
|
@@ -80,13 +66,15 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
80
66
|
get browserPool() {
|
|
81
67
|
return this.#browserPoolDep.value;
|
|
82
68
|
}
|
|
83
|
-
launchContext;
|
|
84
69
|
ignoreShadowRoots;
|
|
85
70
|
ignoreIframes;
|
|
86
71
|
#navigationTimeoutMillis;
|
|
87
72
|
#preNavigationHooks;
|
|
88
73
|
#postNavigationHooks;
|
|
89
74
|
#saveResponseCookies;
|
|
75
|
+
/**
|
|
76
|
+
* @internal
|
|
77
|
+
*/
|
|
90
78
|
static optionsShape = {
|
|
91
79
|
...BasicCrawler.optionsShape,
|
|
92
80
|
navigationTimeoutSecs: schemas.anyNumber
|
|
@@ -94,7 +82,6 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
94
82
|
.default(60),
|
|
95
83
|
preNavigationHooks: schemas.anyArray.default(() => []),
|
|
96
84
|
postNavigationHooks: schemas.anyArray.default(() => []),
|
|
97
|
-
launchContext: schemas.anyObject.default(() => ({})),
|
|
98
85
|
browserPool: validators.browserPool.optional(),
|
|
99
86
|
browserPoolBuilder: schemas.anyFunction.optional(),
|
|
100
87
|
remoteBrowser: schemas.anyObject.optional(),
|
|
@@ -103,18 +90,18 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
103
90
|
ignoreIframes: z.boolean().default(false),
|
|
104
91
|
ignoreShadowRoots: z.boolean().default(false),
|
|
105
92
|
};
|
|
93
|
+
/** @internal */
|
|
106
94
|
static optionsSchema = z.strictObject(BrowserCrawler.optionsShape);
|
|
107
95
|
/**
|
|
108
96
|
* All `BrowserCrawler` parameters are passed via an options object.
|
|
97
|
+
* @internal
|
|
109
98
|
*/
|
|
110
99
|
constructor(options) {
|
|
111
|
-
const { navigationTimeoutSecs, saveResponseCookies,
|
|
100
|
+
const { navigationTimeoutSecs, saveResponseCookies, browserPool, remoteBrowser, preNavigationHooks, postNavigationHooks, ignoreIframes, ignoreShadowRoots, contextPipelineBuilder, browserPoolBuilder, extendContext, ...basicCrawlerOptions } = parseArgument(options, BrowserCrawler.optionsSchema, 'BrowserCrawlerOptions');
|
|
112
101
|
if (browserPool) {
|
|
113
102
|
assertBrowserPoolNotConfigured(new.target.name, { remoteBrowser });
|
|
114
103
|
}
|
|
115
|
-
const skipGuard = (action) => ({
|
|
116
|
-
action: async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {})),
|
|
117
|
-
});
|
|
104
|
+
const skipGuard = (action) => async (ctx) => ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {});
|
|
118
105
|
super({
|
|
119
106
|
...basicCrawlerOptions,
|
|
120
107
|
contextPipelineBuilder: () => {
|
|
@@ -129,7 +116,7 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
129
116
|
}
|
|
130
117
|
return addTimeoutToPromise(async () => step(ctx), remaining, `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
131
118
|
});
|
|
132
|
-
let pipeline = contextPipelineBuilder().compose(
|
|
119
|
+
let pipeline = contextPipelineBuilder().compose(this.#prepareNavigation.bind(this));
|
|
133
120
|
for (const hook of this.#preNavigationHooks) {
|
|
134
121
|
pipeline = pipeline.compose(windowGuard(hook));
|
|
135
122
|
}
|
|
@@ -138,13 +125,12 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
138
125
|
pipeline = pipeline.compose(windowGuard(hook));
|
|
139
126
|
}
|
|
140
127
|
return pipeline
|
|
141
|
-
.compose(skipGuard(this
|
|
142
|
-
.compose(
|
|
143
|
-
.compose(
|
|
128
|
+
.compose(skipGuard(this.#finalizeNavigation.bind(this)))
|
|
129
|
+
.compose(this.#handleBlockedRequestByContent.bind(this))
|
|
130
|
+
.compose(this.#restoreRequestState.bind(this));
|
|
144
131
|
},
|
|
145
132
|
extendContext,
|
|
146
133
|
});
|
|
147
|
-
this.launchContext = launchContext;
|
|
148
134
|
this.#navigationTimeoutMillis = navigationTimeoutSecs * 1000;
|
|
149
135
|
// The public option hooks are extension-aware; internal storage uses the base context type
|
|
150
136
|
// (the pipeline composes hooks against the concrete context, which does not statically carry
|
|
@@ -156,43 +142,32 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
156
142
|
this.#saveResponseCookies = saveResponseCookies;
|
|
157
143
|
this.#browserPoolDep = OwnedOrInjected.resolve(browserPool, () => browserPoolBuilder(remoteBrowser));
|
|
158
144
|
}
|
|
145
|
+
/** @internal */
|
|
159
146
|
getNavigationTimeoutMillis() {
|
|
160
147
|
return this.#navigationTimeoutMillis;
|
|
161
148
|
}
|
|
162
149
|
buildContextPipeline() {
|
|
163
|
-
return ContextPipeline.create().compose(
|
|
164
|
-
action: this.preparePage.bind(this),
|
|
165
|
-
cleanup: async (context) => {
|
|
166
|
-
context.registerDeferredCleanup(async () => {
|
|
167
|
-
const error = !context.session.isUsable()
|
|
168
|
-
? new SessionError('Session is no longer usable')
|
|
169
|
-
: undefined;
|
|
170
|
-
await this.browserPool
|
|
171
|
-
.closePage(context.page, { error })
|
|
172
|
-
.catch((closeError) => this.log.debug('Error while closing page', { error: closeError }));
|
|
173
|
-
});
|
|
174
|
-
},
|
|
175
|
-
});
|
|
150
|
+
return ContextPipeline.create().compose(this.#preparePage.bind(this));
|
|
176
151
|
}
|
|
177
|
-
async containsSelectors(page, selectors) {
|
|
152
|
+
async #containsSelectors(page, selectors) {
|
|
178
153
|
const foundSelectors = (await Promise.all(selectors.map((selector) => page.$(selector))))
|
|
179
154
|
.map((x, i) => [x, selectors[i]])
|
|
180
155
|
.filter(([x]) => x !== null)
|
|
181
156
|
.map(([, selector]) => selector);
|
|
182
157
|
return foundSelectors.length > 0 ? foundSelectors : null;
|
|
183
158
|
}
|
|
184
|
-
async isRequestBlocked(crawlingContext) {
|
|
159
|
+
async #isRequestBlocked(crawlingContext) {
|
|
185
160
|
const { page, response } = crawlingContext;
|
|
186
161
|
// Cloudflare specific heuristic - wait 5 seconds if we get a 403 for the JS challenge to load / resolve.
|
|
187
|
-
if ((await this
|
|
162
|
+
if ((await this.#containsSelectors(page, CLOUDFLARE_RETRY_CSS_SELECTORS)) && response?.status() === 403) {
|
|
188
163
|
await sleep(5000);
|
|
189
164
|
// here we cannot test for response code, because we only have the original response, not the possible Cloudflare redirect on passed challenge.
|
|
190
|
-
const foundSelectors = await this
|
|
165
|
+
const foundSelectors = await this.#containsSelectors(page, RETRY_CSS_SELECTORS);
|
|
191
166
|
if (!foundSelectors)
|
|
192
167
|
return false;
|
|
193
168
|
return `Cloudflare challenge failed, found selectors: ${foundSelectors.join(', ')}`;
|
|
194
169
|
}
|
|
195
|
-
const foundSelectors = await this
|
|
170
|
+
const foundSelectors = await this.#containsSelectors(page, RETRY_CSS_SELECTORS);
|
|
196
171
|
const statusCode = response?.status() ?? 0;
|
|
197
172
|
if (foundSelectors)
|
|
198
173
|
return `Found selectors: ${foundSelectors.join(', ')}`;
|
|
@@ -200,12 +175,34 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
200
175
|
return `Received blocked status code: ${statusCode}`;
|
|
201
176
|
return false;
|
|
202
177
|
}
|
|
203
|
-
async preparePage(crawlingContext) {
|
|
178
|
+
async #preparePage(crawlingContext, onCleanup) {
|
|
204
179
|
const page = await this.browserPool.newPage({
|
|
205
180
|
id: crawlingContext.id,
|
|
206
181
|
session: crawlingContext.session,
|
|
207
182
|
});
|
|
208
183
|
tryCancel();
|
|
184
|
+
onCleanup((failure) => {
|
|
185
|
+
crawlingContext.registerDeferredCleanup(async () => {
|
|
186
|
+
const cause = failure instanceof RequestHandlerError || failure instanceof ContextPipelineInitializationError
|
|
187
|
+
? failure.cause
|
|
188
|
+
: failure;
|
|
189
|
+
// Only tells the pool what to do with the page's browser state - the request's failure, if any, has
|
|
190
|
+
// already been handled. A thrown `SessionError` or a blocked session counts as a block; a session
|
|
191
|
+
// unusable for any other reason is reported as retired. `retire()`/`markBad()` also bump the usage
|
|
192
|
+
// count, so the block check must come first.
|
|
193
|
+
const { session } = crawlingContext;
|
|
194
|
+
const closeReason = cause instanceof SessionError
|
|
195
|
+
? cause
|
|
196
|
+
: session.isBlocked()
|
|
197
|
+
? new SessionError()
|
|
198
|
+
: session.isUsable()
|
|
199
|
+
? undefined
|
|
200
|
+
: new SessionRetiredError();
|
|
201
|
+
await this.browserPool
|
|
202
|
+
.closePage(page, { error: closeReason })
|
|
203
|
+
.catch((closeError) => this.log.debug('Error while closing page', { error: closeError }));
|
|
204
|
+
});
|
|
205
|
+
});
|
|
209
206
|
const addRequests = crawlingContext.addRequests;
|
|
210
207
|
const extractLinks = async (options) => {
|
|
211
208
|
return extractUrlsFromPage(page, options?.selector ?? 'a', options?.baseUrl ?? crawlingContext.request.loadedUrl ?? crawlingContext.request.url);
|
|
@@ -235,16 +232,19 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
235
232
|
},
|
|
236
233
|
};
|
|
237
234
|
}
|
|
238
|
-
async prepareNavigation(crawlingContext) {
|
|
235
|
+
async #prepareNavigation(crawlingContext) {
|
|
239
236
|
if (crawlingContext.request.skipNavigation) {
|
|
240
237
|
return {
|
|
241
238
|
request: new Proxy(crawlingContext.request, {
|
|
242
|
-
get(target, propertyName
|
|
239
|
+
get(target, propertyName) {
|
|
243
240
|
if (propertyName === 'loadedUrl') {
|
|
244
241
|
throw new NavigationSkippedError('The `request.loadedUrl` property is not available - `skipNavigation` was used');
|
|
245
242
|
}
|
|
246
|
-
|
|
243
|
+
// `target` as receiver and bound methods, or `#` members throw on the proxy
|
|
244
|
+
const value = Reflect.get(target, propertyName, target);
|
|
245
|
+
return typeof value === 'function' ? value.bind(target) : value;
|
|
247
246
|
},
|
|
247
|
+
set: (target, propertyName, value) => Reflect.set(target, propertyName, value, target),
|
|
248
248
|
}),
|
|
249
249
|
get response() {
|
|
250
250
|
throw new NavigationSkippedError('The `response` property is not available - `skipNavigation` was used');
|
|
@@ -259,6 +259,7 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
259
259
|
[COOKIES_BEFORE_HOOKS]: this.getCookieHeaderFromRequest(crawlingContext.request),
|
|
260
260
|
};
|
|
261
261
|
}
|
|
262
|
+
// oxlint-disable-next-line crawlee/prefer-private-fields -- patched by @crawlee/otel
|
|
262
263
|
async navigate(crawlingContext) {
|
|
263
264
|
tryCancel();
|
|
264
265
|
const gotoOptions = crawlingContext.gotoOptions;
|
|
@@ -276,22 +277,22 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
276
277
|
}
|
|
277
278
|
const cookiesBeforeHooks = readContextField(crawlingContext, COOKIES_BEFORE_HOOKS);
|
|
278
279
|
const cookiesAfterHooks = this.getCookieHeaderFromRequest(crawlingContext.request);
|
|
279
|
-
await this
|
|
280
|
+
await this.#applyCookies(crawlingContext, cookiesBeforeHooks, cookiesAfterHooks);
|
|
280
281
|
let response;
|
|
281
282
|
try {
|
|
282
283
|
response = (await this.navigationHandler(crawlingContext, gotoOptions)) ?? undefined;
|
|
283
284
|
}
|
|
284
285
|
catch (error) {
|
|
285
|
-
await this
|
|
286
|
+
await this.#handleNavigationTimeout(crawlingContext, error);
|
|
286
287
|
crawlingContext.request.state = RequestState.ERROR;
|
|
287
|
-
this
|
|
288
|
+
this.#throwIfProxyError(error);
|
|
288
289
|
throw error;
|
|
289
290
|
}
|
|
290
291
|
tryCancel();
|
|
291
292
|
crawlingContext.request.state = RequestState.AFTER_NAV;
|
|
292
293
|
return { response };
|
|
293
294
|
}
|
|
294
|
-
async finalizeNavigation(crawlingContext) {
|
|
295
|
+
async #finalizeNavigation(crawlingContext) {
|
|
295
296
|
tryCancel();
|
|
296
297
|
let response;
|
|
297
298
|
try {
|
|
@@ -301,17 +302,17 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
301
302
|
// `preparePage` installs a throwing getter for `response`; reaching this branch means
|
|
302
303
|
// navigation produced no response and no hook overrode it. Treat as undefined.
|
|
303
304
|
}
|
|
304
|
-
await this
|
|
305
|
+
await this.#processResponse(response, crawlingContext);
|
|
305
306
|
tryCancel();
|
|
306
307
|
// Persist cookies from the navigation response before the user handler runs.
|
|
307
308
|
// Cookies set during `requestHandler` are saved again afterwards.
|
|
308
|
-
await this
|
|
309
|
+
await this.#persistCookiesFromPage(crawlingContext);
|
|
309
310
|
return { request: crawlingContext.request };
|
|
310
311
|
}
|
|
311
312
|
/**
|
|
312
313
|
* Copies cookies from the live browser page into the session cookie jar.
|
|
313
314
|
*/
|
|
314
|
-
async persistCookiesFromPage(crawlingContext) {
|
|
315
|
+
async #persistCookiesFromPage(crawlingContext) {
|
|
315
316
|
if (!this.#saveResponseCookies || !crawlingContext.session) {
|
|
316
317
|
return;
|
|
317
318
|
}
|
|
@@ -333,6 +334,7 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
333
334
|
/**
|
|
334
335
|
* Runs the user request handler, then re-reads browser cookies so login flows /
|
|
335
336
|
* `page.setCookie` / XHR `Set-Cookie` updates are stored for later requests.
|
|
337
|
+
* @internal
|
|
336
338
|
*/
|
|
337
339
|
async runRequestHandler(crawlingContext) {
|
|
338
340
|
try {
|
|
@@ -341,7 +343,7 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
341
343
|
finally {
|
|
342
344
|
if (!crawlingContext.request.skipNavigation) {
|
|
343
345
|
try {
|
|
344
|
-
await this
|
|
346
|
+
await this.#persistCookiesFromPage(crawlingContext);
|
|
345
347
|
}
|
|
346
348
|
catch {
|
|
347
349
|
// Page may already be closed on some failure paths; ignore.
|
|
@@ -349,19 +351,19 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
349
351
|
}
|
|
350
352
|
}
|
|
351
353
|
}
|
|
352
|
-
async handleBlockedRequestByContent(crawlingContext) {
|
|
354
|
+
async #handleBlockedRequestByContent(crawlingContext) {
|
|
353
355
|
if (this.retryOnBlocked) {
|
|
354
|
-
const error = await this
|
|
356
|
+
const error = await this.#isRequestBlocked(crawlingContext);
|
|
355
357
|
if (error)
|
|
356
358
|
throw new SessionError(error);
|
|
357
359
|
}
|
|
358
360
|
return {};
|
|
359
361
|
}
|
|
360
|
-
async restoreRequestState(crawlingContext) {
|
|
362
|
+
async #restoreRequestState(crawlingContext) {
|
|
361
363
|
crawlingContext.request.state = RequestState.REQUEST_HANDLER;
|
|
362
364
|
return {};
|
|
363
365
|
}
|
|
364
|
-
async applyCookies({ session, request, page }, preHooksCookies, postHooksCookies) {
|
|
366
|
+
async #applyCookies({ session, request, page }, preHooksCookies, postHooksCookies) {
|
|
365
367
|
const sessionCookie = session
|
|
366
368
|
? (await session.cookieJar.getCookies(request.url)).map(toughCookieToBrowserPoolCookie)
|
|
367
369
|
: [];
|
|
@@ -375,7 +377,7 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
375
377
|
/**
|
|
376
378
|
* Marks session bad on navigation timeout, and stops in-flight page loading on any navigation error.
|
|
377
379
|
*/
|
|
378
|
-
async handleNavigationTimeout(crawlingContext, error) {
|
|
380
|
+
async #handleNavigationTimeout(crawlingContext, error) {
|
|
379
381
|
const { session, page } = crawlingContext;
|
|
380
382
|
// Fire-and-forget: no user code will run on this page after a failed navigation.
|
|
381
383
|
// Swallow rejections: the page may already be detached.
|
|
@@ -390,12 +392,12 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
390
392
|
/**
|
|
391
393
|
* Transforms proxy-related errors to `SessionError`.
|
|
392
394
|
*/
|
|
393
|
-
throwIfProxyError(error) {
|
|
395
|
+
#throwIfProxyError(error) {
|
|
394
396
|
if (this.isProxyError(error)) {
|
|
395
397
|
throw new SessionError(this.getMessageFromError(error));
|
|
396
398
|
}
|
|
397
399
|
}
|
|
398
|
-
async processResponse(response, crawlingContext) {
|
|
400
|
+
async #processResponse(response, crawlingContext) {
|
|
399
401
|
const { session, request, page } = crawlingContext;
|
|
400
402
|
if (typeof response === 'object' && typeof response.status === 'function') {
|
|
401
403
|
const status = response.status();
|
|
@@ -427,13 +429,17 @@ export class BrowserCrawler extends BasicCrawler {
|
|
|
427
429
|
request.loadedUrl = await page.url();
|
|
428
430
|
}
|
|
429
431
|
/**
|
|
430
|
-
*
|
|
431
|
-
*
|
|
432
|
+
* Closes the browsers of a pool the crawler owns, so a finished run leaves none behind. The pool itself is
|
|
433
|
+
* crawler-lifetime and survives — destroying it here would hand a repeated `run()` a dead pool.
|
|
432
434
|
*/
|
|
433
435
|
async teardown() {
|
|
434
|
-
await this.#browserPoolDep.ifOwned((pool) => pool.
|
|
436
|
+
await this.#browserPoolDep.ifOwned((pool) => pool.releaseAllBrowsers());
|
|
435
437
|
await super.teardown();
|
|
436
438
|
}
|
|
439
|
+
async destroy() {
|
|
440
|
+
await super.destroy();
|
|
441
|
+
await this.#browserPoolDep.ifOwned((pool) => pool.destroy());
|
|
442
|
+
}
|
|
437
443
|
}
|
|
438
444
|
/**
|
|
439
445
|
* Extracts URLs from a given page.
|
|
@@ -72,18 +72,19 @@ export interface BrowserLaunchContext<TOptions, Launcher> extends BrowserPluginO
|
|
|
72
72
|
* the browser they run against is only known to the concrete `*BrowserPool()` factory, which is where the
|
|
73
73
|
* caller-facing types are pinned down.
|
|
74
74
|
*/
|
|
75
|
-
|
|
75
|
+
type LauncherBrowserPoolOptions = Omit<BrowserPoolOptions, 'browserPlugins'> & {
|
|
76
76
|
[Hook in keyof BrowserPoolHooks<any, any, any>]?: readonly ((...args: any[]) => unknown)[];
|
|
77
77
|
};
|
|
78
78
|
/**
|
|
79
79
|
* The {@link RemoteBrowserPool} counterpart of {@link LauncherBrowserPoolOptions}.
|
|
80
80
|
*/
|
|
81
|
-
|
|
81
|
+
type LauncherRemoteBrowserPoolOptions = Omit<RemoteBrowserPoolOptions, 'browserPlugins'>;
|
|
82
82
|
/**
|
|
83
83
|
* Abstract class for creating browser launchers, such as `PlaywrightLauncher` and `PuppeteerLauncher`.
|
|
84
84
|
* @ignore
|
|
85
85
|
*/
|
|
86
86
|
export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Launcher = Plugin['library'], T extends Constructor<Plugin> = Constructor<Plugin>, LaunchOptions extends Dictionary<any> | undefined = Partial<Parameters<Plugin['launch']>[0]>, LaunchResult extends ReturnType<Plugin['launch']> = ReturnType<Plugin['launch']>> {
|
|
87
|
+
#private;
|
|
87
88
|
readonly configuration: Configuration;
|
|
88
89
|
launcher: Launcher;
|
|
89
90
|
proxyUrl?: string;
|
|
@@ -92,6 +93,9 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
92
93
|
otherLaunchContextProps: Dictionary;
|
|
93
94
|
Plugin: T;
|
|
94
95
|
userAgent?: string;
|
|
96
|
+
/**
|
|
97
|
+
* @internal
|
|
98
|
+
*/
|
|
95
99
|
protected static optionsShape: {
|
|
96
100
|
proxyUrl: z.ZodOptional<z.ZodURL>;
|
|
97
101
|
useChrome: z.ZodOptional<z.ZodBoolean>;
|
|
@@ -102,6 +106,7 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
102
106
|
launchOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
103
107
|
userAgent: z.ZodOptional<z.ZodString>;
|
|
104
108
|
};
|
|
109
|
+
/** @internal */
|
|
105
110
|
protected static optionsSchema: z.ZodObject<{
|
|
106
111
|
proxyUrl: z.ZodOptional<z.ZodURL>;
|
|
107
112
|
useChrome: z.ZodOptional<z.ZodBoolean>;
|
|
@@ -136,11 +141,6 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
136
141
|
* @internal
|
|
137
142
|
*/
|
|
138
143
|
createRemoteBrowserPool<Page>(options: LauncherRemoteBrowserPoolOptions): RemoteBrowserPool<Page>;
|
|
139
|
-
/**
|
|
140
|
-
* A custom `userAgent` and Crawlee's fingerprint injection would both write the same headers, so an
|
|
141
|
-
* explicitly requested user agent wins.
|
|
142
|
-
*/
|
|
143
|
-
private resolveFingerprinting;
|
|
144
144
|
/**
|
|
145
145
|
* Launches a browser instance based on the plugin.
|
|
146
146
|
* @returns Browser instance.
|
|
@@ -148,10 +148,5 @@ export declare abstract class BrowserLauncher<Plugin extends BrowserPlugin, Laun
|
|
|
148
148
|
launch(): LaunchResult;
|
|
149
149
|
createLaunchOptions(): Dictionary;
|
|
150
150
|
protected getDefaultHeadlessOption(): boolean;
|
|
151
|
-
private getChromeExecutablePath;
|
|
152
|
-
/**
|
|
153
|
-
* Gets a typical path to Chrome executable, depending on the current operating system.
|
|
154
|
-
*/
|
|
155
|
-
private getTypicalChromeExecutablePath;
|
|
156
|
-
private validateProxyUrlProtocol;
|
|
157
151
|
}
|
|
152
|
+
export {};
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import fs from 'node:fs';
|
|
2
2
|
import { createRequire } from 'node:module';
|
|
3
3
|
import os from 'node:os';
|
|
4
|
-
import { Configuration,
|
|
4
|
+
import { Configuration, serviceLocator } from '@crawlee/basic';
|
|
5
5
|
import { BrowserPool, RemoteBrowserPool } from '@crawlee/browser-pool';
|
|
6
|
+
import { schemas } from '@crawlee/utils/internal';
|
|
6
7
|
import { z } from 'zod';
|
|
7
8
|
const DEFAULT_VIEWPORT = {
|
|
8
9
|
width: 1366,
|
|
@@ -23,6 +24,9 @@ export class BrowserLauncher {
|
|
|
23
24
|
// to be provided by child classes;
|
|
24
25
|
Plugin;
|
|
25
26
|
userAgent;
|
|
27
|
+
/**
|
|
28
|
+
* @internal
|
|
29
|
+
*/
|
|
26
30
|
static optionsShape = {
|
|
27
31
|
proxyUrl: z.url().optional(),
|
|
28
32
|
useChrome: z.boolean().optional(),
|
|
@@ -33,6 +37,7 @@ export class BrowserLauncher {
|
|
|
33
37
|
launchOptions: schemas.anyObject.optional(),
|
|
34
38
|
userAgent: z.string().optional(),
|
|
35
39
|
};
|
|
40
|
+
/** @internal */
|
|
36
41
|
static optionsSchema = z.strictObject(BrowserLauncher.optionsShape);
|
|
37
42
|
static requireLauncherOrThrow(launcher, apifyImageName) {
|
|
38
43
|
try {
|
|
@@ -56,7 +61,7 @@ export class BrowserLauncher {
|
|
|
56
61
|
constructor(launchContext, configuration = Configuration.getGlobalConfiguration()) {
|
|
57
62
|
this.configuration = configuration;
|
|
58
63
|
const { launcher, proxyUrl, useChrome, userAgent, launchOptions = {}, ...otherLaunchContextProps } = launchContext;
|
|
59
|
-
this
|
|
64
|
+
this.#validateProxyUrlProtocol(proxyUrl);
|
|
60
65
|
// those need to be reassigned otherwise they are {} in types
|
|
61
66
|
this.launcher = launcher;
|
|
62
67
|
this.proxyUrl = proxyUrl;
|
|
@@ -86,7 +91,7 @@ export class BrowserLauncher {
|
|
|
86
91
|
// parameter, so the argument cannot be checked here. The concrete `*BrowserPool()` factories are where the
|
|
87
92
|
// caller-facing hook types get pinned down.
|
|
88
93
|
return new BrowserPool({
|
|
89
|
-
...this
|
|
94
|
+
...this.#resolveFingerprinting(options),
|
|
90
95
|
browserPlugins: [this.createBrowserPlugin()],
|
|
91
96
|
});
|
|
92
97
|
}
|
|
@@ -99,14 +104,14 @@ export class BrowserLauncher {
|
|
|
99
104
|
return new RemoteBrowserPool({
|
|
100
105
|
...options,
|
|
101
106
|
browserPlugins: [this.createBrowserPlugin()],
|
|
102
|
-
browserPoolOptions: this
|
|
107
|
+
browserPoolOptions: this.#resolveFingerprinting(options.browserPoolOptions ?? {}),
|
|
103
108
|
});
|
|
104
109
|
}
|
|
105
110
|
/**
|
|
106
111
|
* A custom `userAgent` and Crawlee's fingerprint injection would both write the same headers, so an
|
|
107
112
|
* explicitly requested user agent wins.
|
|
108
113
|
*/
|
|
109
|
-
resolveFingerprinting(options) {
|
|
114
|
+
#resolveFingerprinting(options) {
|
|
110
115
|
if (!this.userAgent) {
|
|
111
116
|
return options;
|
|
112
117
|
}
|
|
@@ -143,20 +148,20 @@ export class BrowserLauncher {
|
|
|
143
148
|
launchOptions.headless = this.getDefaultHeadlessOption();
|
|
144
149
|
}
|
|
145
150
|
if (this.useChrome && !launchOptions.executablePath) {
|
|
146
|
-
launchOptions.executablePath = this
|
|
151
|
+
launchOptions.executablePath = this.#getChromeExecutablePath();
|
|
147
152
|
}
|
|
148
153
|
return launchOptions;
|
|
149
154
|
}
|
|
150
155
|
getDefaultHeadlessOption() {
|
|
151
156
|
return this.configuration.headless && !this.configuration.xvfb;
|
|
152
157
|
}
|
|
153
|
-
getChromeExecutablePath() {
|
|
154
|
-
return this.configuration.chromeExecutablePath ?? this
|
|
158
|
+
#getChromeExecutablePath() {
|
|
159
|
+
return this.configuration.chromeExecutablePath ?? this.#getTypicalChromeExecutablePath();
|
|
155
160
|
}
|
|
156
161
|
/**
|
|
157
162
|
* Gets a typical path to Chrome executable, depending on the current operating system.
|
|
158
163
|
*/
|
|
159
|
-
getTypicalChromeExecutablePath() {
|
|
164
|
+
#getTypicalChromeExecutablePath() {
|
|
160
165
|
/**
|
|
161
166
|
* Returns path of Chrome executable by its OS environment variable to deal with non-english language OS.
|
|
162
167
|
* Taking also into account the old [chrome 380177 issue](https://bugs.chromium.org/p/chromium/issues/detail?id=380177).
|
|
@@ -184,7 +189,7 @@ export class BrowserLauncher {
|
|
|
184
189
|
return '/usr/bin/google-chrome';
|
|
185
190
|
}
|
|
186
191
|
}
|
|
187
|
-
validateProxyUrlProtocol(proxyUrl) {
|
|
192
|
+
#validateProxyUrlProtocol(proxyUrl) {
|
|
188
193
|
if (!proxyUrl)
|
|
189
194
|
return;
|
|
190
195
|
if (!/^(http|https|socks4|socks5)/i.test(proxyUrl)) {
|
package/package.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@crawlee/browser",
|
|
3
|
-
"version": "4.0.0-rc.
|
|
3
|
+
"version": "4.0.0-rc.1",
|
|
4
4
|
"description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
|
|
5
5
|
"engines": {
|
|
6
|
-
"node": ">=22.
|
|
6
|
+
"node": ">=22.13.0"
|
|
7
7
|
},
|
|
8
8
|
"type": "module",
|
|
9
9
|
"exports": {
|
|
@@ -47,14 +47,13 @@
|
|
|
47
47
|
"access": "public"
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@apify/timeout": "^0.
|
|
51
|
-
"@crawlee/basic": "4.0.0-rc.
|
|
52
|
-
"@crawlee/browser-pool": "4.0.0-rc.
|
|
53
|
-
"@crawlee/types": "4.0.0-rc.
|
|
54
|
-
"@crawlee/utils": "4.0.0-rc.
|
|
50
|
+
"@apify/timeout": "^1.0.1",
|
|
51
|
+
"@crawlee/basic": "4.0.0-rc.1",
|
|
52
|
+
"@crawlee/browser-pool": "4.0.0-rc.1",
|
|
53
|
+
"@crawlee/types": "4.0.0-rc.1",
|
|
54
|
+
"@crawlee/utils": "4.0.0-rc.1",
|
|
55
55
|
"tslib": "^2.8.1",
|
|
56
|
-
"
|
|
57
|
-
"zod": "^4.4.3"
|
|
56
|
+
"zod": "^4.5.4"
|
|
58
57
|
},
|
|
59
58
|
"peerDependencies": {
|
|
60
59
|
"playwright": "*",
|
|
@@ -75,5 +74,5 @@
|
|
|
75
74
|
}
|
|
76
75
|
}
|
|
77
76
|
},
|
|
78
|
-
"gitHead": "
|
|
77
|
+
"gitHead": "f354ca5e943bed1a5c657a1fce1974c053f9fff0"
|
|
79
78
|
}
|