@crawlee/puppeteer 4.0.0-beta.13 → 4.0.0-beta.131

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,45 +1,48 @@
1
- import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, GetUserDataFromRequest, RouterRoutes } from '@crawlee/browser';
2
- import { BrowserCrawler, Configuration } from '@crawlee/browser';
3
- import type { PuppeteerController, PuppeteerPlugin } from '@crawlee/browser-pool';
1
+ import type { BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook, GetUserDataFromRequest, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/browser';
2
+ import { BrowserCrawler } from '@crawlee/browser';
4
3
  import type { Dictionary } from '@crawlee/types';
5
4
  // @ts-ignore optional peer dependency or compatibility with es2022
6
5
  import type { HTTPResponse, LaunchOptions, Page } from 'puppeteer';
6
+ import { z } from 'zod';
7
+ import type { EnqueueLinksByClickingElementsOptions } from './enqueue-links/click-elements.js';
7
8
  import type { PuppeteerLaunchContext } from './puppeteer-launcher.js';
8
- import type { DirectNavigationOptions, PuppeteerContextUtils } from './utils/puppeteer_utils.js';
9
- export interface PuppeteerCrawlingContext<UserData extends Dictionary = Dictionary> extends BrowserCrawlingContext<Page, HTTPResponse, PuppeteerController, UserData>, PuppeteerContextUtils {
9
+ import type { InterceptHandler } from './utils/puppeteer_request_interception.js';
10
+ import type { BlockRequestsOptions, DirectNavigationOptions, InfiniteScrollOptions, InjectFileOptions, PuppeteerContextUtils, SaveSnapshotOptions } from './utils/puppeteer_utils.js';
11
+ export type PuppeteerGoToOptions = NonNullable<Parameters<Page['goto']>[1]>;
12
+ export interface PuppeteerCrawlingContext<UserData extends Dictionary = any> extends BrowserCrawlingContext<Page, HTTPResponse, UserData, PuppeteerGoToOptions>, PuppeteerContextUtils {
10
13
  }
11
- // @ts-ignore optional peer dependency or compatibility with es2022
12
- export interface PuppeteerHook extends BrowserHook<PuppeteerCrawlingContext, PuppeteerGoToOptions> {
13
- }
14
- export type PuppeteerGoToOptions = Parameters<Page['goto']>[1];
15
- export interface PuppeteerCrawlerOptions<ContextExtension = {}, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension> extends BrowserCrawlerOptions<Page, HTTPResponse, PuppeteerController, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, {
16
- browserPlugins: [PuppeteerPlugin];
17
- }> {
14
+ export type PuppeteerHook<UserData extends Dictionary = any> = BrowserHook<PuppeteerCrawlingContext<UserData>>;
15
+ export interface PuppeteerCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawlerOptions<Page, HTTPResponse, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
18
16
  /**
19
17
  * Options used by {@link launchPuppeteer} to start new Puppeteer instances.
20
18
  */
21
19
  launchContext?: PuppeteerLaunchContext;
20
+ /**
21
+ * Whether to run browser in headless mode. Defaults to `true`.
22
+ * Can be also set via {@link Configuration}.
23
+ */
24
+ headless?: boolean | 'new' | 'old';
22
25
  /**
23
26
  * Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
24
- * or browser properties before navigation. The function accepts two parameters, `crawlingContext` and `gotoOptions`,
25
- * which are passed to the `page.goto()` function the crawler calls to navigate.
27
+ * or browser properties before navigation. The function receives the `crawlingContext`; the options object
28
+ * forwarded to `page.goto()` is available as `crawlingContext.gotoOptions` and can be mutated in place.
29
+ * A hook may optionally return a partial object whose properties are merged into the crawling context
30
+ * (e.g. to override context members for subsequent hooks and pipeline stages).
26
31
  * Example:
27
32
  * ```
28
33
  * preNavigationHooks: [
29
- * async (crawlingContext, gotoOptions) => {
30
- * const { page } = crawlingContext;
34
+ * async ({ page, gotoOptions }) => {
31
35
  * await page.evaluate((attr) => { window.foo = attr; }, 'bar');
36
+ * gotoOptions.timeout = 60_000;
32
37
  * },
33
38
  * ]
34
39
  * ```
35
- *
36
- * Modyfing `pageOptions` is supported only in Playwright incognito.
37
- * See {@link PrePageCreateHook}
38
40
  */
39
- preNavigationHooks?: PuppeteerHook[];
41
+ preNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
40
42
  /**
41
43
  * Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
42
- * The function accepts `crawlingContext` as the only parameter.
44
+ * The function accepts `crawlingContext` as the only parameter. A hook may optionally return a partial object
45
+ * whose properties are merged into the crawling context (e.g. to override `response` after solving a challenge).
43
46
  * Example:
44
47
  * ```
45
48
  * postNavigationHooks: [
@@ -52,7 +55,7 @@ export interface PuppeteerCrawlerOptions<ContextExtension = {}, ExtendedContext
52
55
  * ]
53
56
  * ```
54
57
  */
55
- postNavigationHooks?: PuppeteerHook[];
58
+ postNavigationHooks?: BrowserHook<PuppeteerCrawlingContext<GetUserDataFromRequest<ExtendedContext['request']>>, ContextExtension>[];
56
59
  }
57
60
  /**
58
61
  * Provides a simple framework for parallel crawling of web pages
@@ -65,24 +68,26 @@ export interface PuppeteerCrawlerOptions<ContextExtension = {}, ExtendedContext
65
68
  * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
66
69
  * which downloads the pages using raw HTTP requests and is about 10x faster.
67
70
  *
68
- * The source URLs are represented using {@link Request} objects that are fed from
69
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
70
- * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
71
+ * The source URLs are represented using {@link Request} objects that are fed from the
72
+ * {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
73
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
74
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
75
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
76
+ * result as `requestManager`.
71
77
  *
72
- * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
73
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
74
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
78
+ * > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
79
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
75
80
  *
76
81
  * The crawler finishes when there are no more {@link Request} objects to crawl.
77
82
  *
78
83
  * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
79
84
  * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
80
85
  *
81
- * New pages are only opened when there is enough free CPU and memory available,
82
- * using the functionality provided by the {@link AutoscaledPool} class.
83
- * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
84
- * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
85
- * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
86
+ * New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
87
+ * {@link ConcurrencySystem}.
88
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
89
+ * `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
90
+ * {@link ConcurrencySystem|`concurrencySystem`}.
86
91
  *
87
92
  * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
88
93
  *
@@ -117,90 +122,147 @@ export interface PuppeteerCrawlerOptions<ContextExtension = {}, ExtendedContext
117
122
  * ```
118
123
  * @category Crawlers
119
124
  */
120
- export declare class PuppeteerCrawler<ContextExtension = {}, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension> extends BrowserCrawler<Page, HTTPResponse, PuppeteerController, {
121
- browserPlugins: [PuppeteerPlugin];
122
- }, LaunchOptions, PuppeteerCrawlingContext, ContextExtension, ExtendedContext> {
123
- readonly config: Configuration;
125
+ export declare class PuppeteerCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends PuppeteerCrawlingContext = PuppeteerCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<PuppeteerCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends BrowserCrawler<Page, HTTPResponse, LaunchOptions, PuppeteerCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
124
126
  protected static optionsShape: {
125
- // @ts-ignore optional peer dependency or compatibility with es2022
126
- browserPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
127
- // @ts-ignore optional peer dependency or compatibility with es2022
128
- navigationTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
129
- // @ts-ignore optional peer dependency or compatibility with es2022
130
- preNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
131
- // @ts-ignore optional peer dependency or compatibility with es2022
132
- postNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
133
- // @ts-ignore optional peer dependency or compatibility with es2022
134
- launchContext: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
135
- // @ts-ignore optional peer dependency or compatibility with es2022
136
- headless: import("ow").AnyPredicate<string | boolean>;
137
- // @ts-ignore optional peer dependency or compatibility with es2022
138
- sessionPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
139
- // @ts-ignore optional peer dependency or compatibility with es2022
140
- persistCookiesPerSession: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
141
- // @ts-ignore optional peer dependency or compatibility with es2022
142
- useSessionPool: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
143
- // @ts-ignore optional peer dependency or compatibility with es2022
144
- proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
145
- // @ts-ignore optional peer dependency or compatibility with es2022
146
- contextPipelineBuilder: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
147
- // @ts-ignore optional peer dependency or compatibility with es2022
148
- extendContext: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
149
- // @ts-ignore optional peer dependency or compatibility with es2022
150
- requestList: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
151
- // @ts-ignore optional peer dependency or compatibility with es2022
152
- requestQueue: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
153
- // @ts-ignore optional peer dependency or compatibility with es2022
154
- requestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
155
- // @ts-ignore optional peer dependency or compatibility with es2022
156
- requestHandlerTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
157
- // @ts-ignore optional peer dependency or compatibility with es2022
158
- errorHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
159
- // @ts-ignore optional peer dependency or compatibility with es2022
160
- failedRequestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
161
- // @ts-ignore optional peer dependency or compatibility with es2022
162
- maxRequestRetries: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
163
- // @ts-ignore optional peer dependency or compatibility with es2022
164
- sameDomainDelaySecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
165
- // @ts-ignore optional peer dependency or compatibility with es2022
166
- maxSessionRotations: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
167
- // @ts-ignore optional peer dependency or compatibility with es2022
168
- maxRequestsPerCrawl: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
169
- // @ts-ignore optional peer dependency or compatibility with es2022
170
- autoscaledPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
171
- // @ts-ignore optional peer dependency or compatibility with es2022
172
- statusMessageLoggingInterval: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
173
- // @ts-ignore optional peer dependency or compatibility with es2022
174
- statusMessageCallback: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
175
- // @ts-ignore optional peer dependency or compatibility with es2022
176
- retryOnBlocked: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
177
- // @ts-ignore optional peer dependency or compatibility with es2022
178
- respectRobotsTxtFile: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
179
- // @ts-ignore optional peer dependency or compatibility with es2022
180
- onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
181
- // @ts-ignore optional peer dependency or compatibility with es2022
182
- httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
183
- // @ts-ignore optional peer dependency or compatibility with es2022
184
- minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
185
- // @ts-ignore optional peer dependency or compatibility with es2022
186
- maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
187
- // @ts-ignore optional peer dependency or compatibility with es2022
188
- maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
189
- // @ts-ignore optional peer dependency or compatibility with es2022
190
- keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
191
- // @ts-ignore optional peer dependency or compatibility with es2022
192
- log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
193
- // @ts-ignore optional peer dependency or compatibility with es2022
194
- experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
195
- // @ts-ignore optional peer dependency or compatibility with es2022
196
- statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
127
+ headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
128
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
129
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
130
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
131
+ launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
132
+ browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
133
+ browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
134
+ remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
135
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
136
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
137
+ ignoreIframes: z.ZodDefault<z.ZodBoolean>;
138
+ ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
139
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
140
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
141
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
142
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
143
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
144
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
145
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
146
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
147
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
148
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
149
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
150
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
151
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
152
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
153
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
154
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
155
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
156
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
157
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
158
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
159
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
160
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
161
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
162
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
163
+ requestQueue: z.ZodOptional<z.ZodEnum<{
164
+ deferred: "deferred";
165
+ writeThrough: "writeThrough";
166
+ }>>;
167
+ }, z.core.$strict>]>>;
168
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
169
+ // @ts-ignore optional peer dependency or compatibility with es2022
170
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
171
+ // @ts-ignore optional peer dependency or compatibility with es2022
172
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").Configuration, import("@crawlee/browser").Configuration>>;
173
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
174
+ // @ts-ignore optional peer dependency or compatibility with es2022
175
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").EventManager, import("@crawlee/browser").EventManager>>;
176
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
177
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
178
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
179
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
180
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
181
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
182
+ id: z.ZodOptional<z.ZodString>;
197
183
  };
184
+ protected static optionsSchema: z.ZodObject<{
185
+ headless: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodString]>>;
186
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
187
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
188
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
189
+ launchContext: z.ZodDefault<z.ZodCustom<Dictionary, Dictionary>>;
190
+ browserPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
191
+ browserPoolBuilder: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
192
+ remoteBrowser: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
193
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
194
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
195
+ ignoreIframes: z.ZodDefault<z.ZodBoolean>;
196
+ ignoreShadowRoots: z.ZodDefault<z.ZodBoolean>;
197
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
198
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
199
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
200
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
201
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
202
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
203
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
204
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
205
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
206
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
207
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
208
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
209
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
210
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
211
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
212
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
213
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
214
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
215
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
216
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
217
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
218
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
219
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
220
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
221
+ requestQueue: z.ZodOptional<z.ZodEnum<{
222
+ deferred: "deferred";
223
+ writeThrough: "writeThrough";
224
+ }>>;
225
+ }, z.core.$strict>]>>;
226
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
227
+ // @ts-ignore optional peer dependency or compatibility with es2022
228
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
229
+ // @ts-ignore optional peer dependency or compatibility with es2022
230
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").Configuration, import("@crawlee/browser").Configuration>>;
231
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
232
+ // @ts-ignore optional peer dependency or compatibility with es2022
233
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/browser").EventManager, import("@crawlee/browser").EventManager>>;
234
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
235
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
236
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
237
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
238
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
239
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
240
+ id: z.ZodOptional<z.ZodString>;
241
+ }, z.core.$strict>;
198
242
  /**
199
243
  * All `PuppeteerCrawler` parameters are passed via an options object.
200
244
  */
201
- constructor(options?: PuppeteerCrawlerOptions<ContextExtension, ExtendedContext>, config?: Configuration);
245
+ constructor(options?: PuppeteerCrawlerOptions<ContextExtension, ExtendedContext, Routes, StatisticStateExtension>);
246
+ // @ts-ignore optional peer dependency or compatibility with es2022
247
+ protected buildContextPipeline(): import("@crawlee/browser").ContextPipeline<import("@crawlee/browser").CrawlingContext<Dictionary>, BrowserCrawlingContext<Page, HTTPResponse, Dictionary, Dictionary> & {
248
+ injectFile: (filePath: string, options?: InjectFileOptions) => Promise<unknown>;
249
+ injectJQuery: () => Promise<void>;
250
+ waitForSelector: (selector: string, timeoutMs?: number) => Promise<void>;
251
+ // @ts-ignore optional peer dependency or compatibility with es2022
252
+ parseWithCheerio: (selector?: string, timeoutMs?: number) => Promise<import("@crawlee/browser").CheerioAPI>;
253
+ // @ts-ignore optional peer dependency or compatibility with es2022
254
+ enqueueLinksByClickingElements: (options: Omit<EnqueueLinksByClickingElementsOptions, "page" | "requestManager">) => Promise<import("@crawlee/types").BatchAddRequestsResult>;
255
+ blockRequests: (options?: BlockRequestsOptions) => Promise<void>;
256
+ // @ts-ignore optional peer dependency or compatibility with es2022
257
+ compileScript: (scriptString: string, ctx?: Dictionary) => import("./utils/puppeteer_utils.js").CompiledScriptFunction;
258
+ addInterceptRequestHandler: (handler: InterceptHandler) => Promise<void>;
259
+ removeInterceptRequestHandler: (handler: InterceptHandler) => Promise<void>;
260
+ infiniteScroll: (options?: InfiniteScrollOptions) => Promise<void>;
261
+ saveSnapshot: (options?: SaveSnapshotOptions) => Promise<void>;
262
+ closeCookieModals: () => Promise<void>;
263
+ }>;
202
264
  private enhanceContext;
203
- protected _navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions): Promise<HTTPResponse | null>;
265
+ protected navigationHandler(crawlingContext: PuppeteerCrawlingContext, gotoOptions: DirectNavigationOptions): Promise<HTTPResponse | null>;
204
266
  }
205
267
  /**
206
268
  * Creates new {@link Router} instance that works based on request labels.
@@ -226,6 +288,6 @@ export declare class PuppeteerCrawler<ContextExtension = {}, ExtendedContext ext
226
288
  * await crawler.run();
227
289
  * ```
228
290
  */
229
- // @ts-ignore optional peer dependency or compatibility with es2022
230
- export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData>): import("@crawlee/browser").RouterHandler<Context>;
231
- //# sourceMappingURL=puppeteer-crawler.d.ts.map
291
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
292
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
293
+ export declare function createPuppeteerRouter<Context extends PuppeteerCrawlingContext = PuppeteerCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
@@ -1,6 +1,7 @@
1
- import { BrowserCrawler, Configuration, RequestState, Router } from '@crawlee/browser';
2
- import ow from 'ow';
3
- import { PuppeteerLauncher } from './puppeteer-launcher.js';
1
+ import { assertBrowserPoolNotConfigured, BrowserCrawler, RequestState, Router } from '@crawlee/browser';
2
+ import { parseArgument, serviceLocator } from '@crawlee/core';
3
+ import { z } from 'zod';
4
+ import { puppeteerBrowserPool, remotePuppeteerBrowserPool } from './puppeteer-browser-pool.js';
4
5
  import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js';
5
6
  /**
6
7
  * Provides a simple framework for parallel crawling of web pages
@@ -13,24 +14,26 @@ import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js';
13
14
  * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
14
15
  * which downloads the pages using raw HTTP requests and is about 10x faster.
15
16
  *
16
- * The source URLs are represented using {@link Request} objects that are fed from
17
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
18
- * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
17
+ * The source URLs are represented using {@link Request} objects that are fed from the
18
+ * {@link IRequestManager|request manager} provided via the {@link PuppeteerCrawlerOptions.requestManager|`requestManager`}
19
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
20
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
21
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
22
+ * result as `requestManager`.
19
23
  *
20
- * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
21
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
22
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
24
+ * > The {@link PuppeteerCrawlerOptions.requestList|`requestList`} and {@link PuppeteerCrawlerOptions.requestQueue|`requestQueue`}
25
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
23
26
  *
24
27
  * The crawler finishes when there are no more {@link Request} objects to crawl.
25
28
  *
26
29
  * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
27
30
  * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.requestHandler} option.
28
31
  *
29
- * New pages are only opened when there is enough free CPU and memory available,
30
- * using the functionality provided by the {@link AutoscaledPool} class.
31
- * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
32
- * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
33
- * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
32
+ * New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
33
+ * {@link ConcurrencySystem}.
34
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
35
+ * `PuppeteerCrawler` constructor, or, for finer control, by injecting a pre-configured
36
+ * {@link ConcurrencySystem|`concurrencySystem`}.
34
37
  *
35
38
  * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
36
39
  *
@@ -66,43 +69,43 @@ import { gotoExtended, puppeteerUtils } from './utils/puppeteer_utils.js';
66
69
  * @category Crawlers
67
70
  */
68
71
  export class PuppeteerCrawler extends BrowserCrawler {
69
- config;
70
72
  static optionsShape = {
71
73
  ...BrowserCrawler.optionsShape,
72
- browserPoolOptions: ow.optional.object,
74
+ // Deliberately looser than the declared type: Puppeteer's own accepted string values have moved over
75
+ // time (`'new'`/`'old'`, now `'shell'`), and the value is forwarded to it verbatim.
76
+ headless: z.union([z.boolean(), z.string()]).optional(),
73
77
  };
78
+ static optionsSchema = z.strictObject(PuppeteerCrawler.optionsShape);
74
79
  /**
75
80
  * All `PuppeteerCrawler` parameters are passed via an options object.
76
81
  */
77
- constructor(options = {}, config = Configuration.getGlobalConfig()) {
78
- ow(options, 'PuppeteerCrawlerOptions', ow.object.exactShape(PuppeteerCrawler.optionsShape));
79
- const { launchContext = {}, headless, proxyConfiguration, ...browserCrawlerOptions } = options;
80
- const browserPoolOptions = {
81
- ...options.browserPoolOptions,
82
- };
82
+ constructor(options = {}) {
83
+ const parsedOptions = parseArgument(options, PuppeteerCrawler.optionsSchema, 'PuppeteerCrawlerOptions');
84
+ const { launchContext, headless, configuration, proxyConfiguration, contextPipelineBuilder, ...browserCrawlerOptions } = parsedOptions;
83
85
  if (launchContext.proxyUrl) {
84
86
  throw new Error('PuppeteerCrawlerOptions.launchContext.proxyUrl is not allowed in PuppeteerCrawler.' +
85
87
  'Use PuppeteerCrawlerOptions.proxyConfiguration');
86
88
  }
87
- // `browserPlugins` is working when it's not overridden by `launchContext`,
88
- // which for crawlers it is always overridden. Hence the error to use the other option.
89
- if (browserPoolOptions.browserPlugins) {
90
- throw new Error('browserPoolOptions.browserPlugins is disallowed. Use launchContext.launcher instead.');
91
- }
92
- if (headless != null) {
93
- launchContext.launchOptions ??= {};
94
- launchContext.launchOptions.headless = headless;
89
+ if (options.browserPool) {
90
+ // The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set.
91
+ assertBrowserPoolNotConfigured(new.target.name, {
92
+ launchContext: options.launchContext,
93
+ headless: options.headless,
94
+ });
95
95
  }
96
- const puppeteerLauncher = new PuppeteerLauncher(launchContext, config);
97
- browserPoolOptions.browserPlugins = [puppeteerLauncher.createBrowserPlugin()];
98
96
  super({
99
97
  ...browserCrawlerOptions,
100
98
  launchContext,
99
+ configuration,
101
100
  proxyConfiguration,
102
- browserPoolOptions,
103
- contextPipelineBuilder: () => this.buildContextPipeline().compose({ action: this.enhanceContext.bind(this) }),
104
- }, config);
105
- this.config = config;
101
+ browserPoolBuilder: (remoteBrowser) => remoteBrowser
102
+ ? remotePuppeteerBrowserPool({ ...remoteBrowser, launchContext, headless, configuration })
103
+ : puppeteerBrowserPool({ launchContext, headless, configuration }),
104
+ contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()),
105
+ });
106
+ }
107
+ buildContextPipeline() {
108
+ return super.buildContextPipeline().compose({ action: this.enhanceContext.bind(this) });
106
109
  }
107
110
  async enhanceContext(context) {
108
111
  const waitForSelector = async (selector, timeoutMs = 5_000) => {
@@ -127,7 +130,7 @@ export class PuppeteerCrawler extends BrowserCrawler {
127
130
  },
128
131
  enqueueLinksByClickingElements: async (options) => puppeteerUtils.enqueueLinksByClickingElements({
129
132
  page: context.page,
130
- requestQueue: this.requestQueue,
133
+ requestManager: this.requestManager,
131
134
  ...options,
132
135
  }),
133
136
  blockRequests: async (options) => puppeteerUtils.blockRequests(context.page, options),
@@ -135,39 +138,17 @@ export class PuppeteerCrawler extends BrowserCrawler {
135
138
  addInterceptRequestHandler: async (handler) => puppeteerUtils.addInterceptRequestHandler(context.page, handler),
136
139
  removeInterceptRequestHandler: async (handler) => puppeteerUtils.removeInterceptRequestHandler(context.page, handler),
137
140
  infiniteScroll: async (options) => puppeteerUtils.infiniteScroll(context.page, options),
138
- saveSnapshot: async (options) => puppeteerUtils.saveSnapshot(context.page, { ...options, config: this.config }),
141
+ saveSnapshot: async (options) => puppeteerUtils.saveSnapshot(context.page, {
142
+ ...options,
143
+ configuration: serviceLocator.getConfiguration(),
144
+ }),
139
145
  closeCookieModals: async () => puppeteerUtils.closeCookieModals(context.page),
140
146
  };
141
147
  }
142
- async _navigationHandler(crawlingContext, gotoOptions) {
148
+ async navigationHandler(crawlingContext, gotoOptions) {
143
149
  return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
144
150
  }
145
151
  }
146
- /**
147
- * Creates new {@link Router} instance that works based on request labels.
148
- * This instance can then serve as a `requestHandler` of your {@link PuppeteerCrawler}.
149
- * Defaults to the {@link PuppeteerCrawlingContext}.
150
- *
151
- * > Serves as a shortcut for using `Router.create<PuppeteerCrawlingContext>()`.
152
- *
153
- * ```ts
154
- * import { PuppeteerCrawler, createPuppeteerRouter } from 'crawlee';
155
- *
156
- * const router = createPuppeteerRouter();
157
- * router.addHandler('label-a', async (ctx) => {
158
- * ctx.log.info('...');
159
- * });
160
- * router.addDefaultHandler(async (ctx) => {
161
- * ctx.log.info('...');
162
- * });
163
- *
164
- * const crawler = new PuppeteerCrawler({
165
- * requestHandler: router,
166
- * });
167
- * await crawler.run();
168
- * ```
169
- */
170
- export function createPuppeteerRouter(routes) {
171
- return Router.create(routes);
152
+ export function createPuppeteerRouter(routesOrSchemas) {
153
+ return Router.create(routesOrSchemas);
172
154
  }
173
- //# sourceMappingURL=puppeteer-crawler.js.map