@crawlee/puppeteer 4.0.0-beta.21 → 4.0.0-beta.210
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +2 -4
- package/index.js +1 -3
- package/internals/enqueue-links/click-elements.d.ts +45 -72
- package/internals/enqueue-links/click-elements.js +65 -66
- package/internals/puppeteer-browser-pool.d.ts +55 -0
- package/internals/puppeteer-browser-pool.js +48 -0
- package/internals/puppeteer-crawler.d.ts +163 -118
- package/internals/puppeteer-crawler.js +55 -70
- package/internals/puppeteer-launcher.d.ts +31 -19
- package/internals/puppeteer-launcher.js +18 -14
- package/internals/utils/puppeteer_request_interception.d.ts +0 -1
- package/internals/utils/puppeteer_request_interception.js +8 -8
- package/internals/utils/puppeteer_utils.d.ts +17 -77
- package/internals/utils/puppeteer_utils.js +82 -181
- package/package.json +11 -16
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/enqueue-links/click-elements.d.ts.map +0 -1
- package/internals/enqueue-links/click-elements.js.map +0 -1
- package/internals/puppeteer-crawler.d.ts.map +0 -1
- package/internals/puppeteer-crawler.js.map +0 -1
- package/internals/puppeteer-launcher.d.ts.map +0 -1
- package/internals/puppeteer-launcher.js.map +0 -1
- package/internals/utils/puppeteer_request_interception.d.ts.map +0 -1
- package/internals/utils/puppeteer_request_interception.js.map +0 -1
- package/internals/utils/puppeteer_utils.d.ts.map +0 -1
- package/internals/utils/puppeteer_utils.js.map +0 -1
|
@@ -1,45 +1,46 @@
|
|
|
1
|
-
import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, GetUserDataFromRequest, RouterRoutes } from '@crawlee/browser';
|
|
2
|
-
import { BrowserCrawler
|
|
3
|
-
import type { PuppeteerController, PuppeteerPlugin } from '@crawlee/browser-pool';
|
|
1
|
+
import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, GetUserDataFromRequest, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/browser';
|
|
2
|
+
import { BrowserCrawler } from '@crawlee/browser';
|
|
4
3
|
import type { Dictionary } from '@crawlee/types';
|
|
5
4
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
6
|
-
import type { HTTPResponse,
|
|
5
|
+
import type { HTTPResponse, Page } from 'puppeteer';
|
|
6
|
+
import { z } from 'zod';
|
|
7
7
|
import type { PuppeteerLaunchContext } from './puppeteer-launcher.js';
|
|
8
8
|
import type { DirectNavigationOptions, PuppeteerContextUtils } from './utils/puppeteer_utils.js';
|
|
9
|
-
export
|
|
9
|
+
export type PuppeteerGoToOptions = NonNullable<Parameters<Page['goto']>[1]>;
|
|
10
|
+
export interface PuppeteerCrawlingContext<UserData extends Dictionary = any> extends BrowserCrawlingContext<Page, HTTPResponse, UserData, PuppeteerGoToOptions>, PuppeteerContextUtils {
|
|
10
11
|
}
|
|
11
|
-
|
|
12
|
-
export interface
|
|
13
|
-
}
|
|
14
|
-
export type PuppeteerGoToOptions = Parameters<Page['goto']>[1];
|
|
15
|
-
export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension> extends BrowserCrawlerOptions<Page, HTTPResponse, PuppeteerController, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, {
|
|
16
|
-
browserPlugins: [PuppeteerPlugin];
|
|
17
|
-
}> {
|
|
12
|
+
export type PuppeteerHook<UserData extends Dictionary = any> = BrowserHook<PuppeteerCrawlingContext<UserData>>;
|
|
13
|
+
export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawlerOptions<Page, HTTPResponse, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
18
14
|
/**
|
|
19
15
|
* Options used by {@link launchPuppeteer} to start new Puppeteer instances.
|
|
20
16
|
*/
|
|
21
17
|
launchContext?: PuppeteerLaunchContext;
|
|
18
|
+
/**
|
|
19
|
+
* Whether to run browser in headless mode. Defaults to `true`.
|
|
20
|
+
* Can be also set via {@link Configuration}.
|
|
21
|
+
*/
|
|
22
|
+
headless?: boolean | 'new' | 'old';
|
|
22
23
|
/**
|
|
23
24
|
* Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
|
|
24
|
-
* or browser properties before navigation. The function
|
|
25
|
-
*
|
|
25
|
+
* or browser properties before navigation. The function receives the `crawlingContext`; the options object
|
|
26
|
+
* forwarded to `page.goto()` is available as `crawlingContext.gotoOptions` and can be mutated in place.
|
|
27
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context
|
|
28
|
+
* (e.g. to override context members for subsequent hooks and pipeline stages).
|
|
26
29
|
* Example:
|
|
27
30
|
* ```
|
|
28
31
|
* preNavigationHooks: [
|
|
29
|
-
* async (
|
|
30
|
-
* const { page } = crawlingContext;
|
|
32
|
+
* async ({ page, gotoOptions }) => {
|
|
31
33
|
* await page.evaluate((attr) => { window.foo = attr; }, 'bar');
|
|
34
|
+
* gotoOptions.timeout = 60_000;
|
|
32
35
|
* },
|
|
33
36
|
* ]
|
|
34
37
|
* ```
|
|
35
|
-
*
|
|
36
|
-
* Modyfing `pageOptions` is supported only in Playwright incognito.
|
|
37
|
-
* See {@link PrePageCreateHook}
|
|
38
38
|
*/
|
|
39
|
-
preNavigationHooks?:
|
|
39
|
+
preNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
|
|
40
40
|
/**
|
|
41
41
|
* Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
|
|
42
|
-
* The function accepts `crawlingContext` as the only parameter.
|
|
42
|
+
* The function accepts `crawlingContext` as the only parameter. A hook may optionally return a partial object
|
|
43
|
+
* whose properties are merged into the crawling context (e.g. to override `response` after solving a challenge).
|
|
43
44
|
* Example:
|
|
44
45
|
* ```
|
|
45
46
|
* postNavigationHooks: [
|
|
@@ -52,7 +53,7 @@ export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, E
|
|
|
52
53
|
* ]
|
|
53
54
|
* ```
|
|
54
55
|
*/
|
|
55
|
-
postNavigationHooks?:
|
|
56
|
+
postNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
|
|
56
57
|
}
|
|
57
58
|
/**
|
|
58
59
|
* Provides a simple framework for parallel crawling of web pages
|
|
@@ -65,24 +66,26 @@ export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, E
|
|
|
65
66
|
* If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
|
|
66
67
|
* which downloads the pages using raw HTTP requests and is about 10x faster.
|
|
67
68
|
*
|
|
68
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
69
|
-
* {@link
|
|
70
|
-
*
|
|
69
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
70
|
+
* {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
|
|
71
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
72
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
73
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
74
|
+
* result as `requestManager`.
|
|
71
75
|
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
76
|
+
* > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
|
|
77
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
75
78
|
*
|
|
76
79
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
77
80
|
*
|
|
78
81
|
* `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
|
|
79
82
|
* and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
|
|
80
83
|
*
|
|
81
|
-
* New pages are only opened when there is enough free CPU and memory available,
|
|
82
|
-
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
85
|
-
* {@link
|
|
84
|
+
* New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
|
|
85
|
+
* {@link ConcurrencySystem}.
|
|
86
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
87
|
+
* `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
|
|
88
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
86
89
|
*
|
|
87
90
|
* Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
|
|
88
91
|
*
|
|
@@ -117,94 +120,136 @@ export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, E
|
|
|
117
120
|
* ```
|
|
118
121
|
* @category Crawlers
|
|
119
122
|
*/
|
|
120
|
-
export declare class PuppeteerCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension> extends BrowserCrawler<Page, HTTPResponse,
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
123
|
+
export declare class PuppeteerCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawler<Page, HTTPResponse, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
124
|
+
#private;
|
|
125
|
+
/**
|
|
126
|
+
* @internal
|
|
127
|
+
*/
|
|
124
128
|
protected static optionsShape: {
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
156
|
-
|
|
157
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
|
|
183
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
184
|
-
httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
185
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
186
|
-
minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
187
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
188
|
-
maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
189
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
190
|
-
maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
191
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
192
|
-
keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
193
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
194
|
-
log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
195
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
196
|
-
experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
197
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
198
|
-
statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
199
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
200
|
-
id: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
|
|
129
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
130
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
131
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
132
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
133
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
134
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
135
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
136
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
137
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
138
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
139
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
140
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
141
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
142
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
143
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
144
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
145
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
146
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
147
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
148
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
149
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
150
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
151
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
152
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
153
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
154
|
+
deferred: "deferred";
|
|
155
|
+
writeThrough: "writeThrough";
|
|
156
|
+
}>>;
|
|
157
|
+
}, z.core.$strict>]>>;
|
|
158
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
159
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
160
|
+
httpClient: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/http-client").BaseHttpClient>>;
|
|
161
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
162
|
+
configuration: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/browser").Configuration>>;
|
|
163
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
164
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
165
|
+
eventManager: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/browser").EventManager>>;
|
|
166
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
167
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
168
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
169
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
170
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
171
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
172
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
173
|
+
id: z.ZodOptional<z.ZodString>;
|
|
174
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
175
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
176
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
177
|
+
browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
178
|
+
browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
179
|
+
remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
180
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
181
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
182
|
+
ignoreIframes: z.ZodDefault<z.ZodBoolean>;
|
|
183
|
+
ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
|
|
184
|
+
launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
|
|
185
|
+
headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
|
|
201
186
|
};
|
|
187
|
+
/** @internal */
|
|
188
|
+
protected static optionsSchema: z.ZodObject<{
|
|
189
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
190
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
191
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
192
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
193
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
194
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
195
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
196
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
197
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
198
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
199
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
200
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
201
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
202
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
203
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
204
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
205
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
206
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
207
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
208
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
209
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
210
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
211
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
212
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
213
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
214
|
+
deferred: "deferred";
|
|
215
|
+
writeThrough: "writeThrough";
|
|
216
|
+
}>>;
|
|
217
|
+
}, z.core.$strict>]>>;
|
|
218
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
219
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
220
|
+
httpClient: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/http-client").BaseHttpClient>>;
|
|
221
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
222
|
+
configuration: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/browser").Configuration>>;
|
|
223
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
224
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
225
|
+
eventManager: z.ZodOptional<z.ZodInstanceOf<import("@crawlee/browser").EventManager>>;
|
|
226
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
227
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
228
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
229
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
230
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
231
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
232
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
233
|
+
id: z.ZodOptional<z.ZodString>;
|
|
234
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
235
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
236
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
237
|
+
browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
238
|
+
browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
239
|
+
remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
240
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
241
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
242
|
+
ignoreIframes: z.ZodDefault<z.ZodBoolean>;
|
|
243
|
+
ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
|
|
244
|
+
launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
|
|
245
|
+
headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
|
|
246
|
+
}, z.core.$strict>;
|
|
202
247
|
/**
|
|
203
248
|
* All `PuppeteerCrawler` parameters are passed via an options object.
|
|
204
249
|
*/
|
|
205
|
-
constructor(options?: PuppeteerCrawlerOptions<ContextExtension, ExtendedContext
|
|
250
|
+
constructor(options?: PuppeteerCrawlerOptions<ContextExtension, ExtendedContext, Routes, StatisticStateExtension>);
|
|
206
251
|
private enhanceContext;
|
|
207
|
-
protected
|
|
252
|
+
protected navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions): Promise<HTTPResponse | null>;
|
|
208
253
|
}
|
|
209
254
|
/**
|
|
210
255
|
* Creates new {@link Router} instance that works based on request labels.
|
|
@@ -230,6 +275,6 @@ export declare class PuppeteerCrawler<ContextExtension = Dictionary<never>, Exte
|
|
|
230
275
|
* await crawler.run();
|
|
231
276
|
* ```
|
|
232
277
|
*/
|
|
233
|
-
|
|
234
|
-
export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData
|
|
235
|
-
|
|
278
|
+
export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
279
|
+
export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
280
|
+
export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
@@ -1,7 +1,9 @@
|
|
|
1
|
-
import { BrowserCrawler,
|
|
2
|
-
import
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
1
|
+
import { BrowserCrawler, RequestState, Router } from '@crawlee/browser';
|
|
2
|
+
import { serviceLocator } from '@crawlee/core';
|
|
3
|
+
import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal';
|
|
4
|
+
import { z } from 'zod';
|
|
5
|
+
import { puppeteerBrowserPool, remotePuppeteerBrowserPool } from './puppeteer-browser-pool.js';
|
|
6
|
+
import * as puppeteerUtils from './utils/puppeteer_utils.js';
|
|
5
7
|
/**
|
|
6
8
|
* Provides a simple framework for parallel crawling of web pages
|
|
7
9
|
* using headless Chrome with [Puppeteer](https://github.com/puppeteer/puppeteer).
|
|
@@ -13,24 +15,26 @@ import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js';
|
|
|
13
15
|
* If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
|
|
14
16
|
* which downloads the pages using raw HTTP requests and is about 10x faster.
|
|
15
17
|
*
|
|
16
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
17
|
-
* {@link
|
|
18
|
-
*
|
|
18
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
19
|
+
* {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
|
|
20
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
21
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
22
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
23
|
+
* result as `requestManager`.
|
|
19
24
|
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
25
|
+
* > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
|
|
26
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
23
27
|
*
|
|
24
28
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
25
29
|
*
|
|
26
30
|
* `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
|
|
27
31
|
* and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
|
|
28
32
|
*
|
|
29
|
-
* New pages are only opened when there is enough free CPU and memory available,
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
* {@link
|
|
33
|
+
* New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
|
|
34
|
+
* {@link ConcurrencySystem}.
|
|
35
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
36
|
+
* `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
|
|
37
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
34
38
|
*
|
|
35
39
|
* Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
|
|
36
40
|
*
|
|
@@ -66,43 +70,47 @@ import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js';
|
|
|
66
70
|
* @category Crawlers
|
|
67
71
|
*/
|
|
68
72
|
export class PuppeteerCrawler extends BrowserCrawler {
|
|
69
|
-
|
|
73
|
+
/**
|
|
74
|
+
* @internal
|
|
75
|
+
*/
|
|
70
76
|
static optionsShape = {
|
|
71
77
|
...BrowserCrawler.optionsShape,
|
|
72
|
-
|
|
78
|
+
launchContext: schemas.anyObject.default(() => ({})),
|
|
79
|
+
// Deliberately looser than the declared type: Puppeteer's own accepted string values have moved over
|
|
80
|
+
// time (`'new'`/`'old'`, now `'shell'`), and the value is forwarded to it verbatim.
|
|
81
|
+
headless: z.union([z.boolean(), z.string()]).optional(),
|
|
73
82
|
};
|
|
83
|
+
/** @internal */
|
|
84
|
+
static optionsSchema = z.strictObject(PuppeteerCrawler.optionsShape);
|
|
74
85
|
/**
|
|
75
86
|
* All `PuppeteerCrawler` parameters are passed via an options object.
|
|
76
87
|
*/
|
|
77
|
-
constructor(options = {}
|
|
78
|
-
|
|
79
|
-
const { launchContext
|
|
80
|
-
const browserPoolOptions = {
|
|
81
|
-
...options.browserPoolOptions,
|
|
82
|
-
};
|
|
88
|
+
constructor(options = {}) {
|
|
89
|
+
const parsedOptions = parseArgument(options, PuppeteerCrawler.optionsSchema, 'PuppeteerCrawlerOptions');
|
|
90
|
+
const { launchContext, headless, configuration, proxyConfiguration, ...browserCrawlerOptions } = parsedOptions;
|
|
83
91
|
if (launchContext.proxyUrl) {
|
|
84
92
|
throw new Error('PuppeteerCrawlerOptions.launchContext.proxyUrl is not allowed in PuppeteerCrawler.' +
|
|
85
93
|
'Use PuppeteerCrawlerOptions.proxyConfiguration');
|
|
86
94
|
}
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
95
|
+
if (options.browserPool) {
|
|
96
|
+
// The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set.
|
|
97
|
+
assertBrowserPoolNotConfigured(new.target.name, {
|
|
98
|
+
launchContext: options.launchContext,
|
|
99
|
+
headless: options.headless,
|
|
100
|
+
});
|
|
91
101
|
}
|
|
92
|
-
if (headless != null) {
|
|
93
|
-
launchContext.launchOptions ??= {};
|
|
94
|
-
launchContext.launchOptions.headless = headless;
|
|
95
|
-
}
|
|
96
|
-
const puppeteerLauncher = new PuppeteerLauncher(launchContext, config);
|
|
97
|
-
browserPoolOptions.browserPlugins = [puppeteerLauncher.createBrowserPlugin()];
|
|
98
102
|
super({
|
|
99
103
|
...browserCrawlerOptions,
|
|
100
|
-
|
|
104
|
+
configuration,
|
|
101
105
|
proxyConfiguration,
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
+
browserPoolBuilder: (remoteBrowser) => remoteBrowser
|
|
107
|
+
? remotePuppeteerBrowserPool({ ...remoteBrowser, launchContext, headless, configuration })
|
|
108
|
+
: puppeteerBrowserPool({ launchContext, headless, configuration }),
|
|
109
|
+
contextPipelineBuilder: () => this.#buildContextPipeline(),
|
|
110
|
+
});
|
|
111
|
+
}
|
|
112
|
+
#buildContextPipeline() {
|
|
113
|
+
return this.buildContextPipeline().compose(this.enhanceContext.bind(this));
|
|
106
114
|
}
|
|
107
115
|
async enhanceContext(context) {
|
|
108
116
|
const waitForSelector = async (selector, timeoutMs = 5_000) => {
|
|
@@ -127,7 +135,7 @@ export class PuppeteerCrawler extends BrowserCrawler {
|
|
|
127
135
|
},
|
|
128
136
|
enqueueLinksByClickingElements: async (options) => puppeteerUtils.enqueueLinksByClickingElements({
|
|
129
137
|
page: context.page,
|
|
130
|
-
|
|
138
|
+
requestManager: this.requestManager,
|
|
131
139
|
...options,
|
|
132
140
|
}),
|
|
133
141
|
blockRequests: async (options) => puppeteerUtils.blockRequests(context.page, options),
|
|
@@ -135,39 +143,16 @@ export class PuppeteerCrawler extends BrowserCrawler {
|
|
|
135
143
|
addInterceptRequestHandler: async (handler) => puppeteerUtils.addInterceptRequestHandler(context.page, handler),
|
|
136
144
|
removeInterceptRequestHandler: async (handler) => puppeteerUtils.removeInterceptRequestHandler(context.page, handler),
|
|
137
145
|
infiniteScroll: async (options) => puppeteerUtils.infiniteScroll(context.page, options),
|
|
138
|
-
saveSnapshot: async (options) => puppeteerUtils.saveSnapshot(context.page, {
|
|
139
|
-
|
|
146
|
+
saveSnapshot: async (options) => puppeteerUtils.saveSnapshot(context.page, {
|
|
147
|
+
...options,
|
|
148
|
+
configuration: serviceLocator.getConfiguration(),
|
|
149
|
+
}),
|
|
140
150
|
};
|
|
141
151
|
}
|
|
142
|
-
async
|
|
143
|
-
return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
|
|
152
|
+
async navigationHandler(crawlingContext, gotoOptions) {
|
|
153
|
+
return puppeteerUtils.gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
|
|
144
154
|
}
|
|
145
155
|
}
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
* This instance can then serve as a `requestHandler` of your {@link PuppeteerCrawler}.
|
|
149
|
-
* Defaults to the {@link PuppeteerCrawlingContext}.
|
|
150
|
-
*
|
|
151
|
-
* > Serves as a shortcut for using `Router.create<PuppeteerCrawlingContext>()`.
|
|
152
|
-
*
|
|
153
|
-
* ```ts
|
|
154
|
-
* import { PuppeteerCrawler, createPuppeteerRouter } from 'crawlee';
|
|
155
|
-
*
|
|
156
|
-
* const router = createPuppeteerRouter();
|
|
157
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
158
|
-
* ctx.log.info('...');
|
|
159
|
-
* });
|
|
160
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
161
|
-
* ctx.log.info('...');
|
|
162
|
-
* });
|
|
163
|
-
*
|
|
164
|
-
* const crawler = new PuppeteerCrawler({
|
|
165
|
-
* requestHandler: router,
|
|
166
|
-
* });
|
|
167
|
-
* await crawler.run();
|
|
168
|
-
* ```
|
|
169
|
-
*/
|
|
170
|
-
export function createPuppeteerRouter(routes) {
|
|
171
|
-
return Router.create(routes);
|
|
156
|
+
export function createPuppeteerRouter(routesOrSchemas) {
|
|
157
|
+
return Router.create(routesOrSchemas);
|
|
172
158
|
}
|
|
173
|
-
//# sourceMappingURL=puppeteer-crawler.js.map
|