@crawlee/puppeteer 4.0.0-beta.2 → 4.0.0-beta.200

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,47 +1,46 @@
1
- import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, BrowserRequestHandler, GetUserDataFromRequest, LoadedContext, RouterRoutes } from '@crawlee/browser';
2
- import { BrowserCrawler, Configuration } from '@crawlee/browser';
3
- import type { PuppeteerController, PuppeteerPlugin } from '@crawlee/browser-pool';
1
+ import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, GetUserDataFromRequest, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/browser';
2
+ import { BrowserCrawler } from '@crawlee/browser';
4
3
  import type { Dictionary } from '@crawlee/types';
5
4
  // @ts-ignore optional peer dependency or compatibility with es2022
6
- import type { HTTPResponse, LaunchOptions, Page } from 'puppeteer';
5
+ import type { HTTPResponse, Page } from 'puppeteer';
6
+ import { z } from 'zod';
7
7
  import type { PuppeteerLaunchContext } from './puppeteer-launcher.js';
8
8
  import type { DirectNavigationOptions, PuppeteerContextUtils } from './utils/puppeteer_utils.js';
9
- export interface PuppeteerCrawlingContext<UserData extends Dictionary = Dictionary> extends BrowserCrawlingContext<PuppeteerCrawler, Page, HTTPResponse, PuppeteerController, UserData>, PuppeteerContextUtils {
9
+ export type PuppeteerGoToOptions = NonNullable<Parameters<Page['goto']>[1]>;
10
+ export interface PuppeteerCrawlingContext<UserData extends Dictionary = any> extends BrowserCrawlingContext<Page, HTTPResponse, UserData, PuppeteerGoToOptions>, PuppeteerContextUtils {
10
11
  }
11
- // @ts-ignore optional peer dependency or compatibility with es2022
12
- export interface PuppeteerHook extends BrowserHook<PuppeteerCrawlingContext, PuppeteerGoToOptions> {
13
- }
14
- export interface PuppeteerRequestHandler extends BrowserRequestHandler<LoadedContext<PuppeteerCrawlingContext>> {
15
- }
16
- export type PuppeteerGoToOptions = Parameters<Page['goto']>[1];
17
- export interface PuppeteerCrawlerOptions extends BrowserCrawlerOptions<PuppeteerCrawlingContext, {
18
- browserPlugins: [PuppeteerPlugin];
19
- }> {
12
+ export type PuppeteerHook<UserData extends Dictionary = any> = BrowserHook<PuppeteerCrawlingContext<UserData>>;
13
+ export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawlerOptions<Page, HTTPResponse, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
20
14
  /**
21
15
  * Options used by {@link launchPuppeteer} to start new Puppeteer instances.
22
16
  */
23
17
  launchContext?: PuppeteerLaunchContext;
18
+ /**
19
+ * Whether to run browser in headless mode. Defaults to `true`.
20
+ * Can be also set via {@link Configuration}.
21
+ */
22
+ headless?: boolean | 'new' | 'old';
24
23
  /**
25
24
  * Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
26
- * or browser properties before navigation. The function accepts two parameters, `crawlingContext` and `gotoOptions`,
27
- * which are passed to the `page.goto()` function the crawler calls to navigate.
25
+ * or browser properties before navigation. The function receives the `crawlingContext`; the options object
26
+ * forwarded to `page.goto()` is available as `crawlingContext.gotoOptions` and can be mutated in place.
27
+ * A hook may optionally return a partial object whose properties are merged into the crawling context
28
+ * (e.g. to override context members for subsequent hooks and pipeline stages).
28
29
  * Example:
29
30
  * ```
30
31
  * preNavigationHooks: [
31
- * async (crawlingContext, gotoOptions) => {
32
- * const { page } = crawlingContext;
32
+ * async ({ page, gotoOptions }) => {
33
33
  * await page.evaluate((attr) => { window.foo = attr; }, 'bar');
34
+ * gotoOptions.timeout = 60_000;
34
35
  * },
35
36
  * ]
36
37
  * ```
37
- *
38
- * Modyfing `pageOptions` is supported only in Playwright incognito.
39
- * See {@link PrePageCreateHook}
40
38
  */
41
- preNavigationHooks?: PuppeteerHook[];
39
+ preNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
42
40
  /**
43
41
  * Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
44
- * The function accepts `crawlingContext` as the only parameter.
42
+ * The function accepts `crawlingContext` as the only parameter. A hook may optionally return a partial object
43
+ * whose properties are merged into the crawling context (e.g. to override `response` after solving a challenge).
45
44
  * Example:
46
45
  * ```
47
46
  * postNavigationHooks: [
@@ -54,7 +53,7 @@ export interface PuppeteerCrawlerOptions extends BrowserCrawlerOptions<Puppeteer
54
53
  * ]
55
54
  * ```
56
55
  */
57
- postNavigationHooks?: PuppeteerHook[];
56
+ postNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
58
57
  }
59
58
  /**
60
59
  * Provides a simple framework for parallel crawling of web pages
@@ -67,24 +66,26 @@ export interface PuppeteerCrawlerOptions extends BrowserCrawlerOptions<Puppeteer
67
66
  * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
68
67
  * which downloads the pages using raw HTTP requests and is about 10x faster.
69
68
  *
70
- * The source URLs are represented using {@link Request} objects that are fed from
71
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
72
- * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
69
+ * The source URLs are represented using {@link Request} objects that are fed from the
70
+ * {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
71
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
72
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
73
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
74
+ * result as `requestManager`.
73
75
  *
74
- * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
75
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
76
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
76
+ * > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
77
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
77
78
  *
78
79
  * The crawler finishes when there are no more {@link Request} objects to crawl.
79
80
  *
80
81
  * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
81
82
  * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
82
83
  *
83
- * New pages are only opened when there is enough free CPU and memory available,
84
- * using the functionality provided by the {@link AutoscaledPool} class.
85
- * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
86
- * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
87
- * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
84
+ * New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
85
+ * {@link ConcurrencySystem}.
86
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
87
+ * `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
88
+ * {@link ConcurrencySystem|`concurrencySystem`}.
88
89
  *
89
90
  * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
90
91
  *
@@ -119,91 +120,136 @@ export interface PuppeteerCrawlerOptions extends BrowserCrawlerOptions<Puppeteer
119
120
  * ```
120
121
  * @category Crawlers
121
122
  */
122
- export declare class PuppeteerCrawler extends BrowserCrawler<{
123
- browserPlugins: [PuppeteerPlugin];
124
- }, LaunchOptions, PuppeteerCrawlingContext> {
125
- private readonly options;
126
- readonly config: Configuration;
123
+ export declare class PuppeteerCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawler<Page, HTTPResponse, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
124
+ #private;
125
+ /**
126
+ * @internal
127
+ */
127
128
  protected static optionsShape: {
128
- // @ts-ignore optional peer dependency or compatibility with es2022
129
- browserPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
130
- // @ts-ignore optional peer dependency or compatibility with es2022
131
- navigationTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
132
- // @ts-ignore optional peer dependency or compatibility with es2022
133
- preNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
134
- // @ts-ignore optional peer dependency or compatibility with es2022
135
- postNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
136
- // @ts-ignore optional peer dependency or compatibility with es2022
137
- launchContext: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
138
- // @ts-ignore optional peer dependency or compatibility with es2022
139
- headless: import("ow").AnyPredicate<string | boolean>;
140
- // @ts-ignore optional peer dependency or compatibility with es2022
141
- sessionPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
142
- // @ts-ignore optional peer dependency or compatibility with es2022
143
- persistCookiesPerSession: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
144
- // @ts-ignore optional peer dependency or compatibility with es2022
145
- useSessionPool: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
146
- // @ts-ignore optional peer dependency or compatibility with es2022
147
- proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
148
- // @ts-ignore optional peer dependency or compatibility with es2022
149
- ignoreShadowRoots: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
150
- // @ts-ignore optional peer dependency or compatibility with es2022
151
- ignoreIframes: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
152
- // @ts-ignore optional peer dependency or compatibility with es2022
153
- requestList: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
154
- // @ts-ignore optional peer dependency or compatibility with es2022
155
- requestQueue: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
156
- // @ts-ignore optional peer dependency or compatibility with es2022
157
- requestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
158
- // @ts-ignore optional peer dependency or compatibility with es2022
159
- requestHandlerTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
160
- // @ts-ignore optional peer dependency or compatibility with es2022
161
- errorHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
162
- // @ts-ignore optional peer dependency or compatibility with es2022
163
- failedRequestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
164
- // @ts-ignore optional peer dependency or compatibility with es2022
165
- maxRequestRetries: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
166
- // @ts-ignore optional peer dependency or compatibility with es2022
167
- sameDomainDelaySecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
168
- // @ts-ignore optional peer dependency or compatibility with es2022
169
- maxSessionRotations: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
170
- // @ts-ignore optional peer dependency or compatibility with es2022
171
- maxRequestsPerCrawl: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
172
- // @ts-ignore optional peer dependency or compatibility with es2022
173
- autoscaledPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
174
- // @ts-ignore optional peer dependency or compatibility with es2022
175
- statusMessageLoggingInterval: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
176
- // @ts-ignore optional peer dependency or compatibility with es2022
177
- statusMessageCallback: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
178
- // @ts-ignore optional peer dependency or compatibility with es2022
179
- retryOnBlocked: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
180
- // @ts-ignore optional peer dependency or compatibility with es2022
181
- respectRobotsTxtFile: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
182
- // @ts-ignore optional peer dependency or compatibility with es2022
183
- onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
184
- // @ts-ignore optional peer dependency or compatibility with es2022
185
- httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
186
- // @ts-ignore optional peer dependency or compatibility with es2022
187
- minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
188
- // @ts-ignore optional peer dependency or compatibility with es2022
189
- maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
190
- // @ts-ignore optional peer dependency or compatibility with es2022
191
- maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
192
- // @ts-ignore optional peer dependency or compatibility with es2022
193
- keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
194
- // @ts-ignore optional peer dependency or compatibility with es2022
195
- log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
196
- // @ts-ignore optional peer dependency or compatibility with es2022
197
- experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
198
- // @ts-ignore optional peer dependency or compatibility with es2022
199
- statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
129
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
130
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
131
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
132
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
133
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
134
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
135
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
136
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
137
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
138
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
139
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
140
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
141
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
142
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
143
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
144
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
145
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
146
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
147
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
148
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
149
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
150
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
151
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
152
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
153
+ requestQueue: z.ZodOptional<z.ZodEnum<{
154
+ deferred: "deferred";
155
+ writeThrough: "writeThrough";
156
+ }>>;
157
+ }, z.core.$strict>]>>;
158
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
159
+ // @ts-ignore optional peer dependency or compatibility with es2022
160
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
161
+ // @ts-ignore optional peer dependency or compatibility with es2022
162
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").Configuration, import("@crawlee/browser").Configuration>>;
163
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
164
+ // @ts-ignore optional peer dependency or compatibility with es2022
165
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").EventManager, import("@crawlee/browser").EventManager>>;
166
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
167
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
168
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
169
+ initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
170
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
171
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
172
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
173
+ id: z.ZodOptional<z.ZodString>;
174
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
175
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
176
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
177
+ browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
178
+ browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
179
+ remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
180
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
181
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
182
+ ignoreIframes: z.ZodDefault<z.ZodBoolean>;
183
+ ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
184
+ launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
185
+ headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
200
186
  };
187
+ /** @internal */
188
+ protected static optionsSchema: z.ZodObject<{
189
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
190
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
191
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
192
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
193
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
194
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
195
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
196
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
197
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
198
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
199
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
200
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
201
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
202
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
203
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
204
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
205
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
206
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
207
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
208
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
209
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
210
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
211
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
212
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
213
+ requestQueue: z.ZodOptional<z.ZodEnum<{
214
+ deferred: "deferred";
215
+ writeThrough: "writeThrough";
216
+ }>>;
217
+ }, z.core.$strict>]>>;
218
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
219
+ // @ts-ignore optional peer dependency or compatibility with es2022
220
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
221
+ // @ts-ignore optional peer dependency or compatibility with es2022
222
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").Configuration, import("@crawlee/browser").Configuration>>;
223
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
224
+ // @ts-ignore optional peer dependency or compatibility with es2022
225
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").EventManager, import("@crawlee/browser").EventManager>>;
226
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
227
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
228
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
229
+ initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
230
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
231
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
232
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
233
+ id: z.ZodOptional<z.ZodString>;
234
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
235
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
236
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
237
+ browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
238
+ browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
239
+ remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
240
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
241
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
242
+ ignoreIframes: z.ZodDefault<z.ZodBoolean>;
243
+ ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
244
+ launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
245
+ headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
246
+ }, z.core.$strict>;
201
247
  /**
202
248
  * All `PuppeteerCrawler` parameters are passed via an options object.
203
249
  */
204
- constructor(options?: PuppeteerCrawlerOptions, config?: Configuration);
205
- protected _runRequestHandler(context: PuppeteerCrawlingContext): Promise<void>;
206
- protected _navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions): Promise<HTTPResponse | null>;
250
+ constructor(options?: PuppeteerCrawlerOptions<ContextExtension, ExtendedContext, Routes, StatisticStateExtension>);
251
+ private enhanceContext;
252
+ protected navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions): Promise<HTTPResponse | null>;
207
253
  }
208
254
  /**
209
255
  * Creates new {@link Router} instance that works based on request labels.
@@ -229,6 +275,6 @@ export declare class PuppeteerCrawler extends BrowserCrawler<{
229
275
  * await crawler.run();
230
276
  * ```
231
277
  */
232
- // @ts-ignore optional peer dependency or compatibility with es2022
233
- export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData>): import("@crawlee/browser").RouterHandler<Context>;
234
- //# sourceMappingURL=puppeteer-crawler.d.ts.map
278
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
279
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
280
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
@@ -1,7 +1,9 @@
1
- import { BrowserCrawler, Configuration, Router } from '@crawlee/browser';
2
- import ow from 'ow';
3
- import { PuppeteerLauncher } from './puppeteer-launcher.js';
4
- import { gotoExtended, registerUtilsToContext } from './utils/puppeteer_utils.js';
1
+ import { BrowserCrawler, RequestState, Router } from '@crawlee/browser';
2
+ import { serviceLocator } from '@crawlee/core';
3
+ import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal';
4
+ import { z } from 'zod';
5
+ import { puppeteerBrowserPool, remotePuppeteerBrowserPool } from './puppeteer-browser-pool.js';
6
+ import * as puppeteerUtils from './utils/puppeteer_utils.js';
5
7
  /**
6
8
  * Provides a simple framework for parallel crawling of web pages
7
9
  * using headless Chrome with [Puppeteer](https://github.com/puppeteer/puppeteer).
@@ -13,24 +15,26 @@ import { gotoExtended, registerUtilsToContext } from './utils/puppeteer_utils.js
13
15
  * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
14
16
  * which downloads the pages using raw HTTP requests and is about 10x faster.
15
17
  *
16
- * The source URLs are represented using {@link Request} objects that are fed from
17
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
18
- * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
18
+ * The source URLs are represented using {@link Request} objects that are fed from the
19
+ * {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
20
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
21
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
22
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
23
+ * result as `requestManager`.
19
24
  *
20
- * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
21
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
22
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
25
+ * > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
26
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
23
27
  *
24
28
  * The crawler finishes when there are no more {@link Request} objects to crawl.
25
29
  *
26
30
  * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
27
31
  * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
28
32
  *
29
- * New pages are only opened when there is enough free CPU and memory available,
30
- * using the functionality provided by the {@link AutoscaledPool} class.
31
- * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
32
- * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
33
- * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
33
+ * New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
34
+ * {@link ConcurrencySystem}.
35
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
36
+ * `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
37
+ * {@link ConcurrencySystem|`concurrencySystem`}.
34
38
  *
35
39
  * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
36
40
  *
@@ -66,73 +70,89 @@ import { gotoExtended, registerUtilsToContext } from './utils/puppeteer_utils.js
66
70
  * @category Crawlers
67
71
  */
68
72
  export class PuppeteerCrawler extends BrowserCrawler {
69
- options;
70
- config;
73
+ /**
74
+ * @internal
75
+ */
71
76
  static optionsShape = {
72
77
  ...BrowserCrawler.optionsShape,
73
- browserPoolOptions: ow.optional.object,
78
+ launchContext: schemas.anyObject.default(() => ({})),
79
+ // Deliberately looser than the declared type: Puppeteer's own accepted string values have moved over
80
+ // time (`'new'`/`'old'`, now `'shell'`), and the value is forwarded to it verbatim.
81
+ headless: z.union([z.boolean(), z.string()]).optional(),
74
82
  };
83
+ /** @internal */
84
+ static optionsSchema = z.strictObject(PuppeteerCrawler.optionsShape);
75
85
  /**
76
86
  * All `PuppeteerCrawler` parameters are passed via an options object.
77
87
  */
78
- constructor(options = {}, config = Configuration.getGlobalConfig()) {
79
- ow(options, 'PuppeteerCrawlerOptions', ow.object.exactShape(PuppeteerCrawler.optionsShape));
80
- const { launchContext = {}, headless, proxyConfiguration, ...browserCrawlerOptions } = options;
81
- const browserPoolOptions = {
82
- ...options.browserPoolOptions,
83
- };
88
+ constructor(options = {}) {
89
+ const parsedOptions = parseArgument(options, PuppeteerCrawler.optionsSchema, 'PuppeteerCrawlerOptions');
90
+ const { launchContext, headless, configuration, proxyConfiguration, ...browserCrawlerOptions } = parsedOptions;
84
91
  if (launchContext.proxyUrl) {
85
92
  throw new Error('PuppeteerCrawlerOptions.launchContext.proxyUrl is not allowed in PuppeteerCrawler.' +
86
93
  'Use PuppeteerCrawlerOptions.proxyConfiguration');
87
94
  }
88
- // `browserPlugins` is working when it's not overridden by `launchContext`,
89
- // which for crawlers it is always overridden. Hence the error to use the other option.
90
- if (browserPoolOptions.browserPlugins) {
91
- throw new Error('browserPoolOptions.browserPlugins is disallowed. Use launchContext.launcher instead.');
95
+ if (options.browserPool) {
96
+ // The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set.
97
+ assertBrowserPoolNotConfigured(new.target.name, {
98
+ launchContext: options.launchContext,
99
+ headless: options.headless,
100
+ });
92
101
  }
93
- if (headless != null) {
94
- launchContext.launchOptions ??= {};
95
- launchContext.launchOptions.headless = headless;
96
- }
97
- const puppeteerLauncher = new PuppeteerLauncher(launchContext, config);
98
- browserPoolOptions.browserPlugins = [puppeteerLauncher.createBrowserPlugin()];
99
- super({ ...browserCrawlerOptions, launchContext, proxyConfiguration, browserPoolOptions }, config);
100
- this.options = options;
101
- this.config = config;
102
+ super({
103
+ ...browserCrawlerOptions,
104
+ configuration,
105
+ proxyConfiguration,
106
+ browserPoolBuilder: (remoteBrowser) => remoteBrowser
107
+ ? remotePuppeteerBrowserPool({ ...remoteBrowser, launchContext, headless, configuration })
108
+ : puppeteerBrowserPool({ launchContext, headless, configuration }),
109
+ contextPipelineBuilder: () => this.#buildContextPipeline(),
110
+ });
102
111
  }
103
- async _runRequestHandler(context) {
104
- registerUtilsToContext(context, this.options);
105
- await super._runRequestHandler(context);
112
+ #buildContextPipeline() {
113
+ return this.buildContextPipeline().compose(this.enhanceContext.bind(this));
106
114
  }
107
- async _navigationHandler(crawlingContext, gotoOptions) {
108
- return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
115
+ async enhanceContext(context) {
116
+ const waitForSelector = async (selector, timeoutMs = 5_000) => {
117
+ await context.page.waitForSelector(selector, { timeout: timeoutMs });
118
+ };
119
+ return {
120
+ injectFile: async (filePath, options) => puppeteerUtils.injectFile(context.page, filePath, options),
121
+ injectJQuery: async () => {
122
+ if (context.request.state === RequestState.BEFORE_NAV) {
123
+ context.log.warning('Using injectJQuery() in preNavigationHooks leads to unstable results. Use it in a postNavigationHook or a requestHandler instead.');
124
+ await puppeteerUtils.injectJQuery(context.page);
125
+ return;
126
+ }
127
+ await puppeteerUtils.injectJQuery(context.page, { surviveNavigations: false });
128
+ },
129
+ waitForSelector,
130
+ parseWithCheerio: async (selector, timeoutMs = 5_000) => {
131
+ if (selector) {
132
+ await waitForSelector(selector, timeoutMs);
133
+ }
134
+ return puppeteerUtils.parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes);
135
+ },
136
+ enqueueLinksByClickingElements: async (options) => puppeteerUtils.enqueueLinksByClickingElements({
137
+ page: context.page,
138
+ requestManager: this.requestManager,
139
+ ...options,
140
+ }),
141
+ blockRequests: async (options) => puppeteerUtils.blockRequests(context.page, options),
142
+ compileScript: (scriptString, ctx) => puppeteerUtils.compileScript(scriptString, ctx),
143
+ addInterceptRequestHandler: async (handler) => puppeteerUtils.addInterceptRequestHandler(context.page, handler),
144
+ removeInterceptRequestHandler: async (handler) => puppeteerUtils.removeInterceptRequestHandler(context.page, handler),
145
+ infiniteScroll: async (options) => puppeteerUtils.infiniteScroll(context.page, options),
146
+ saveSnapshot: async (options) => puppeteerUtils.saveSnapshot(context.page, {
147
+ ...options,
148
+ configuration: serviceLocator.getConfiguration(),
149
+ }),
150
+ };
151
+ }
152
+ async navigationHandler(crawlingContext, gotoOptions) {
153
+ return puppeteerUtils.gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
109
154
  }
110
155
  }
111
- /**
112
- * Creates new {@link Router} instance that works based on request labels.
113
- * This instance can then serve as a `requestHandler` of your {@link PuppeteerCrawler}.
114
- * Defaults to the {@link PuppeteerCrawlingContext}.
115
- *
116
- * > Serves as a shortcut for using `Router.create<PuppeteerCrawlingContext>()`.
117
- *
118
- * ```ts
119
- * import { PuppeteerCrawler, createPuppeteerRouter } from 'crawlee';
120
- *
121
- * const router = createPuppeteerRouter();
122
- * router.addHandler('label-a', async (ctx) => {
123
- * ctx.log.info('...');
124
- * });
125
- * router.addDefaultHandler(async (ctx) => {
126
- * ctx.log.info('...');
127
- * });
128
- *
129
- * const crawler = new PuppeteerCrawler({
130
- * requestHandler: router,
131
- * });
132
- * await crawler.run();
133
- * ```
134
- */
135
- export function createPuppeteerRouter(routes) {
136
- return Router.create(routes);
156
+ export function createPuppeteerRouter(routesOrSchemas) {
157
+ return Router.create(routesOrSchemas);
137
158
  }
138
- //# sourceMappingURL=puppeteer-crawler.js.map