@crawlee/playwright 4.0.0-beta.21 → 4.0.0-beta.210

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +14 -14
  2. package/index.d.ts +4 -3
  3. package/index.js +2 -2
  4. package/internals/adaptive-playwright-crawler.d.ts +136 -65
  5. package/internals/adaptive-playwright-crawler.js +417 -274
  6. package/internals/enqueue-links/click-elements.d.ts +36 -64
  7. package/internals/enqueue-links/click-elements.js +65 -67
  8. package/internals/playwright-browser-pool.d.ts +71 -0
  9. package/internals/playwright-browser-pool.js +61 -0
  10. package/internals/playwright-crawler.d.ts +176 -148
  11. package/internals/playwright-crawler.js +77 -73
  12. package/internals/playwright-launcher.d.ts +30 -20
  13. package/internals/playwright-launcher.js +22 -17
  14. package/internals/utils/playwright-utils.d.ts +54 -56
  15. package/internals/utils/playwright-utils.js +117 -138
  16. package/internals/utils/rendering-type-prediction.d.ts +37 -11
  17. package/internals/utils/rendering-type-prediction.js +81 -27
  18. package/package.json +15 -19
  19. package/index.d.ts.map +0 -1
  20. package/index.js.map +0 -1
  21. package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
  22. package/internals/adaptive-playwright-crawler.js.map +0 -1
  23. package/internals/enqueue-links/click-elements.d.ts.map +0 -1
  24. package/internals/enqueue-links/click-elements.js.map +0 -1
  25. package/internals/playwright-crawler.d.ts.map +0 -1
  26. package/internals/playwright-crawler.js.map +0 -1
  27. package/internals/playwright-launcher.d.ts.map +0 -1
  28. package/internals/playwright-launcher.js.map +0 -1
  29. package/internals/utils/playwright-utils.d.ts.map +0 -1
  30. package/internals/utils/playwright-utils.js.map +0 -1
  31. package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
  32. package/internals/utils/rendering-type-prediction.js.map +0 -1
@@ -1,7 +1,8 @@
1
- import { BrowserCrawler, Configuration, RequestState, Router } from '@crawlee/browser';
2
- import ow from 'ow';
3
- import { PlaywrightLauncher } from './playwright-launcher.js';
4
- import { gotoExtended, playwrightUtils } from './utils/playwright-utils.js';
1
+ import { BrowserCrawler, RequestState, Router, serviceLocator } from '@crawlee/browser';
2
+ import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal';
3
+ import { z } from 'zod';
4
+ import { playwrightBrowserPool, remotePlaywrightBrowserPool } from './playwright-browser-pool.js';
5
+ import { blockRequests, compileScript, enqueueLinksByClickingElements, gotoExtended, handleCloudflareChallenge, infiniteScroll, injectFile, injectJQuery, parseWithCheerio, saveSnapshot, } from './utils/playwright-utils.js';
5
6
  /**
6
7
  * Provides a simple framework for parallel crawling of web pages
7
8
  * using headless Chromium, Firefox and Webkit browsers with [Playwright](https://github.com/microsoft/playwright).
@@ -13,24 +14,26 @@ import { gotoExtended, playwrightUtils } from './utils/playwright-utils.js';
13
14
  * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
14
15
  * which downloads the pages using raw HTTP requests and is about 10x faster.
15
16
  *
16
- * The source URLs are represented using {@link Request} objects that are fed from
17
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PlaywrightCrawlerOptions.requestList}
18
- * or {@link PlaywrightCrawlerOptions.requestQueue} constructor options, respectively.
17
+ * The source URLs are represented using {@link Request} objects that are fed from the
18
+ * {@link IRequestManager|request manager} provided via the {@link PlaywrightCrawlerOptions.requestManager|`requestManager`}
19
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
20
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
21
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
22
+ * result as `requestManager`.
19
23
  *
20
- * If both {@link PlaywrightCrawlerOptions.requestList} and {@link PlaywrightCrawlerOptions.requestQueue} are used,
21
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
22
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
24
+ * > The {@link PlaywrightCrawlerOptions.requestList|`requestList`} and {@link PlaywrightCrawlerOptions.requestQueue|`requestQueue`}
25
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
23
26
  *
24
27
  * The crawler finishes when there are no more {@link Request} objects to crawl.
25
28
  *
26
29
  * `PlaywrightCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
27
30
  * and then calls the function provided by user as the {@link PlaywrightCrawlerOptions.requestHandler} option.
28
31
  *
29
- * New pages are only opened when there is enough free CPU and memory available,
30
- * using the functionality provided by the {@link AutoscaledPool} class.
31
- * All {@link AutoscaledPool} configuration options can be passed to the {@link PlaywrightCrawlerOptions.autoscaledPoolOptions}
32
- * parameter of the `PlaywrightCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
33
- * {@link AutoscaledPoolOptions} are available directly in the `PlaywrightCrawler` constructor.
32
+ * New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
33
+ * {@link ConcurrencySystem}.
34
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
35
+ * `PlaywrightCrawler` constructor, or, for finer control, by injecting a pre-configured
36
+ * {@link ConcurrencySystem|`concurrencySystem`}.
34
37
  *
35
38
  * Note that the pool of Playwright instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
36
39
  *
@@ -66,47 +69,46 @@ import { gotoExtended, playwrightUtils } from './utils/playwright-utils.js';
66
69
  * @category Crawlers
67
70
  */
68
71
  export class PlaywrightCrawler extends BrowserCrawler {
69
- config;
72
+ /**
73
+ * @internal
74
+ */
70
75
  static optionsShape = {
71
76
  ...BrowserCrawler.optionsShape,
72
- browserPoolOptions: ow.optional.object,
73
- launcher: ow.optional.object,
74
- ignoreIframes: ow.optional.boolean,
75
- ignoreShadowRoots: ow.optional.boolean,
77
+ launchContext: schemas.anyObject.default(() => ({})),
78
+ headless: z.boolean().optional(),
76
79
  };
80
+ /** @internal */
81
+ static optionsSchema = z.strictObject(PlaywrightCrawler.optionsShape);
77
82
  /**
78
83
  * All `PlaywrightCrawler` parameters are passed via an options object.
79
84
  */
80
- constructor(options = {}, config = Configuration.getGlobalConfig()) {
81
- ow(options, 'PlaywrightCrawlerOptions', ow.object.exactShape(PlaywrightCrawler.optionsShape));
82
- const { launchContext = {}, headless, ...browserCrawlerOptions } = options;
83
- const browserPoolOptions = {
84
- ...options.browserPoolOptions,
85
- };
85
+ constructor(options = {}) {
86
+ const parsedOptions = parseArgument(options, PlaywrightCrawler.optionsSchema, 'PlaywrightCrawlerOptions');
87
+ const { launchContext, headless, configuration, ...browserCrawlerOptions } = parsedOptions;
86
88
  if (launchContext.proxyUrl) {
87
89
  throw new Error('PlaywrightCrawlerOptions.launchContext.proxyUrl is not allowed in PlaywrightCrawler.' +
88
90
  'Use PlaywrightCrawlerOptions.proxyConfiguration');
89
91
  }
90
- // `browserPlugins` is working when it's not overridden by `launchContext`,
91
- // which for crawlers it is always overridden. Hence the error to use the other option.
92
- if (browserPoolOptions.browserPlugins) {
93
- throw new Error('browserPoolOptions.browserPlugins is disallowed. Use launchContext.launcher instead.');
94
- }
95
- if (headless != null) {
96
- launchContext.launchOptions ??= {};
97
- launchContext.launchOptions.headless = headless;
92
+ if (options.browserPool) {
93
+ // The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set.
94
+ assertBrowserPoolNotConfigured(new.target.name, {
95
+ launchContext: options.launchContext,
96
+ headless: options.headless,
97
+ });
98
98
  }
99
- const playwrightLauncher = new PlaywrightLauncher(launchContext, config);
100
- browserPoolOptions.browserPlugins = [playwrightLauncher.createBrowserPlugin()];
101
99
  super({
102
100
  ...browserCrawlerOptions,
103
- launchContext,
104
- browserPoolOptions,
105
- contextPipelineBuilder: () => this.buildContextPipeline().compose({ action: this.enhanceContext.bind(this) }),
106
- }, config);
107
- this.config = config;
101
+ configuration,
102
+ browserPoolBuilder: (remoteBrowser) => remoteBrowser
103
+ ? remotePlaywrightBrowserPool({ ...remoteBrowser, launchContext, headless, configuration })
104
+ : playwrightBrowserPool({ launchContext, headless, configuration }),
105
+ contextPipelineBuilder: () => this.#buildContextPipeline(),
106
+ });
108
107
  }
109
- async _navigationHandler(crawlingContext, gotoOptions) {
108
+ #buildContextPipeline() {
109
+ return this.buildContextPipeline().compose(this.enhanceContext.bind(this));
110
+ }
111
+ async navigationHandler(crawlingContext, gotoOptions) {
110
112
  return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
111
113
  }
112
114
  async enhanceContext(context) {
@@ -114,64 +116,66 @@ export class PlaywrightCrawler extends BrowserCrawler {
114
116
  const locator = context.page.locator(selector).first();
115
117
  await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
116
118
  };
119
+ const downloads = [];
120
+ context.page.on('download', (download) => downloads.push(download));
117
121
  return {
118
- injectFile: async (filePath, options) => playwrightUtils.injectFile(context.page, filePath, options),
122
+ injectFile: async (filePath, options) => injectFile(context.page, filePath, options),
119
123
  injectJQuery: async () => {
120
124
  if (context.request.state === RequestState.BEFORE_NAV) {
121
125
  context.log.warning('Using injectJQuery() in preNavigationHooks leads to unstable results. Use it in a postNavigationHook or a requestHandler instead.');
122
- await playwrightUtils.injectJQuery(context.page);
126
+ await injectJQuery(context.page);
123
127
  return;
124
128
  }
125
- await playwrightUtils.injectJQuery(context.page, { surviveNavigations: false });
129
+ await injectJQuery(context.page, { surviveNavigations: false });
126
130
  },
127
- blockRequests: async (options) => playwrightUtils.blockRequests(context.page, options),
131
+ blockRequests: async (options) => blockRequests(context.page, options),
128
132
  waitForSelector,
129
133
  parseWithCheerio: async (selector, timeoutMs = 5_000) => {
130
134
  if (selector) {
131
135
  await waitForSelector(selector, timeoutMs);
132
136
  }
133
- return playwrightUtils.parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes);
137
+ return parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes);
134
138
  },
135
- infiniteScroll: async (options) => playwrightUtils.infiniteScroll(context.page, options),
136
- saveSnapshot: async (options) => playwrightUtils.saveSnapshot(context.page, { ...options, config: this.config }),
137
- enqueueLinksByClickingElements: async (options) => playwrightUtils.enqueueLinksByClickingElements({
139
+ infiniteScroll: async (options) => infiniteScroll(context.page, options),
140
+ listDownloads: async () => downloads,
141
+ saveSnapshot: async (options) => saveSnapshot(context.page, {
142
+ ...options,
143
+ configuration: serviceLocator.getConfiguration(),
144
+ }),
145
+ enqueueLinksByClickingElements: async (options) => enqueueLinksByClickingElements({
138
146
  ...options,
139
147
  page: context.page,
140
- requestQueue: this.requestQueue,
148
+ requestManager: this.requestManager,
141
149
  }),
142
- compileScript: (scriptString, ctx) => playwrightUtils.compileScript(scriptString, ctx),
143
- closeCookieModals: async () => playwrightUtils.closeCookieModals(context.page),
150
+ compileScript: (scriptString, ctx) => compileScript(scriptString, ctx),
144
151
  handleCloudflareChallenge: async (options) => {
145
- return playwrightUtils.handleCloudflareChallenge(context.page, context.request.url, context.session, options);
152
+ return handleCloudflareChallenge(context.page, context.request.url, options);
146
153
  },
147
154
  };
148
155
  }
149
156
  }
150
157
  /**
151
- * Creates new {@link Router} instance that works based on request labels.
152
- * This instance can then serve as a `requestHandler` of your {@link PlaywrightCrawler}.
153
- * Defaults to the {@link PlaywrightCrawlingContext}.
154
- *
155
- * > Serves as a shortcut for using `Router.create<PlaywrightCrawlingContext>()`.
158
+ * Returns a `postNavigationHooks`-ready hook that runs {@link PlaywrightContextUtils.handleCloudflareChallenge}
159
+ * and propagates the post-challenge {@link Response} back into the crawling context via its return value.
156
160
  *
161
+ * **Example usage**
157
162
  * ```ts
158
- * import { PlaywrightCrawler, createPlaywrightRouter } from 'crawlee';
159
- *
160
- * const router = createPlaywrightRouter();
161
- * router.addHandler('label-a', async (ctx) => {
162
- * ctx.log.info('...');
163
- * });
164
- * router.addDefaultHandler(async (ctx) => {
165
- * ctx.log.info('...');
166
- * });
163
+ * import { PlaywrightCrawler, handleCloudflareChallengeHook } from 'crawlee';
167
164
  *
168
165
  * const crawler = new PlaywrightCrawler({
169
- * requestHandler: router,
166
+ * postNavigationHooks: [handleCloudflareChallengeHook()],
170
167
  * });
171
- * await crawler.run();
172
168
  * ```
173
169
  */
174
- export function createPlaywrightRouter(routes) {
175
- return Router.create(routes);
170
+ export function handleCloudflareChallengeHook(options) {
171
+ return async (context) => {
172
+ const response = await context.handleCloudflareChallenge(options);
173
+ if (response !== undefined) {
174
+ return { response };
175
+ }
176
+ return undefined;
177
+ };
178
+ }
179
+ export function createPlaywrightRouter(routesOrSchemas) {
180
+ return Router.create(routesOrSchemas);
176
181
  }
177
- //# sourceMappingURL=playwright-crawler.js.map
@@ -3,6 +3,7 @@ import { BrowserLauncher, Configuration } from '@crawlee/browser';
3
3
  import { PlaywrightPlugin } from '@crawlee/browser-pool';
4
4
  // @ts-ignore optional peer dependency or compatibility with es2022
5
5
  import type { Browser, BrowserType, LaunchOptions } from 'playwright';
6
+ import { z } from 'zod';
6
7
  /**
7
8
  * Apify extends the launch options of Playwright.
8
9
  * You can use any of the Playwright compatible
@@ -70,31 +71,41 @@ export interface PlaywrightLaunchContext extends BrowserLaunchContext<LaunchOpti
70
71
  * @ignore
71
72
  */
72
73
  export declare class PlaywrightLauncher extends BrowserLauncher<PlaywrightPlugin> {
73
- readonly config: Configuration;
74
+ readonly configuration: Configuration;
75
+ /**
76
+ * @internal
77
+ */
74
78
  protected static optionsShape: {
79
+ proxyUrl: z.ZodOptional<z.ZodURL>;
80
+ useChrome: z.ZodOptional<z.ZodBoolean>;
81
+ useIncognitoPages: z.ZodOptional<z.ZodBoolean>;
82
+ browserPerProxy: z.ZodOptional<z.ZodBoolean>;
83
+ ignoreProxyCertificate: z.ZodOptional<z.ZodBoolean>;
84
+ userDataDir: z.ZodOptional<z.ZodString>;
75
85
  // @ts-ignore optional peer dependency or compatibility with es2022
76
- launcher: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
77
- // @ts-ignore optional peer dependency or compatibility with es2022
78
- launchContextOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
79
- // @ts-ignore optional peer dependency or compatibility with es2022
80
- proxyUrl: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
81
- // @ts-ignore optional peer dependency or compatibility with es2022
82
- useChrome: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
86
+ launchOptions: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
87
+ userAgent: z.ZodOptional<z.ZodString>;
83
88
  // @ts-ignore optional peer dependency or compatibility with es2022
84
- useIncognitoPages: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
85
- // @ts-ignore optional peer dependency or compatibility with es2022
86
- browserPerProxy: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
87
- // @ts-ignore optional peer dependency or compatibility with es2022
88
- userDataDir: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
89
+ launcher: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
90
+ };
91
+ /** @internal */
92
+ protected static optionsSchema: z.ZodObject<{
93
+ proxyUrl: z.ZodOptional<z.ZodURL>;
94
+ useChrome: z.ZodOptional<z.ZodBoolean>;
95
+ useIncognitoPages: z.ZodOptional<z.ZodBoolean>;
96
+ browserPerProxy: z.ZodOptional<z.ZodBoolean>;
97
+ ignoreProxyCertificate: z.ZodOptional<z.ZodBoolean>;
98
+ userDataDir: z.ZodOptional<z.ZodString>;
89
99
  // @ts-ignore optional peer dependency or compatibility with es2022
90
- launchOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
100
+ launchOptions: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
101
+ userAgent: z.ZodOptional<z.ZodString>;
91
102
  // @ts-ignore optional peer dependency or compatibility with es2022
92
- userAgent: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
93
- };
103
+ launcher: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
104
+ }, z.core.$strict>;
94
105
  /**
95
106
  * All `PlaywrightLauncher` parameters are passed via this launchContext object.
96
107
  */
97
- constructor(launchContext?: PlaywrightLaunchContext, config?: Configuration);
108
+ constructor(launchContext?: PlaywrightLaunchContext, configuration?: Configuration);
98
109
  }
99
110
  /**
100
111
  * Launches headless browsers using Playwright pre-configured to work within the Apify platform.
@@ -125,9 +136,8 @@ export declare class PlaywrightLauncher extends BrowserLauncher<PlaywrightPlugin
125
136
  * Optional settings passed to `browserType.launch()`. In addition to
126
137
  * [Playwright's options](https://playwright.dev/docs/api/class-browsertype?_highlight=launch#browsertypelaunchoptions)
127
138
  * the object may contain our own {@link PlaywrightLaunchContext} that enable additional features.
128
- * @param [config]
139
+ * @param [configuration]
129
140
  * @returns
130
141
  * Promise that resolves to Playwright's `Browser` instance.
131
142
  */
132
- export declare function launchPlaywright(launchContext?: PlaywrightLaunchContext, config?: Configuration): Promise<Browser>;
133
- //# sourceMappingURL=playwright-launcher.d.ts.map
143
+ export declare function launchPlaywright(launchContext?: PlaywrightLaunchContext, configuration?: Configuration): Promise<Browser>;
@@ -1,33 +1,39 @@
1
1
  import { BrowserLauncher, Configuration } from '@crawlee/browser';
2
2
  import { PlaywrightPlugin } from '@crawlee/browser-pool';
3
- import ow from 'ow';
3
+ import { parseArgument, schemas } from '@crawlee/utils/internal';
4
+ import { z } from 'zod';
4
5
  /**
5
6
  * `PlaywrightLauncher` is based on the `BrowserLauncher`. It launches `playwright` browser instance.
6
7
  * @ignore
7
8
  */
8
9
  export class PlaywrightLauncher extends BrowserLauncher {
9
- config;
10
+ configuration;
11
+ /**
12
+ * @internal
13
+ */
10
14
  static optionsShape = {
11
15
  ...BrowserLauncher.optionsShape,
12
- launcher: ow.optional.object,
13
- launchContextOptions: ow.optional.object,
16
+ // Passthrough schema — the launcher module object must keep its prototype through parsing.
17
+ launcher: schemas.anyObject.optional(),
14
18
  };
19
+ /** @internal */
20
+ static optionsSchema = z.strictObject(PlaywrightLauncher.optionsShape);
15
21
  /**
16
22
  * All `PlaywrightLauncher` parameters are passed via this launchContext object.
17
23
  */
18
- constructor(launchContext = {}, config = Configuration.getGlobalConfig()) {
19
- ow(launchContext, 'PlaywrightLauncherOptions', ow.object.exactShape(PlaywrightLauncher.optionsShape));
20
- const { launcher = BrowserLauncher.requireLauncherOrThrow('playwright', 'apify/actor-node-playwright-*').chromium, } = launchContext;
21
- const { launchOptions = {}, ...rest } = launchContext;
24
+ constructor(launchContext = {}, configuration = Configuration.getGlobalConfiguration()) {
25
+ const parsedContext = parseArgument(launchContext, PlaywrightLauncher.optionsSchema, 'PlaywrightLaunchContext');
26
+ const { launcher = BrowserLauncher.requireLauncherOrThrow('playwright', 'apify/actor-node-playwright-*').chromium, } = parsedContext;
27
+ const { launchOptions = {}, ...rest } = parsedContext;
22
28
  super({
23
29
  ...rest,
24
30
  launchOptions: {
25
31
  ...launchOptions,
26
- executablePath: getDefaultExecutablePath(launchContext, config),
32
+ executablePath: getDefaultExecutablePath(parsedContext, configuration),
27
33
  },
28
34
  launcher,
29
- }, config);
30
- this.config = config;
35
+ }, configuration);
36
+ this.configuration = configuration;
31
37
  this.Plugin = PlaywrightPlugin;
32
38
  }
33
39
  }
@@ -36,8 +42,8 @@ export class PlaywrightLauncher extends BrowserLauncher {
36
42
  * @returns default path to browser.
37
43
  * @ignore
38
44
  */
39
- function getDefaultExecutablePath(launchContext, config) {
40
- const pathFromPlaywrightImage = config.get('defaultBrowserPath');
45
+ function getDefaultExecutablePath(launchContext, configuration) {
46
+ const pathFromPlaywrightImage = configuration.defaultBrowserPath;
41
47
  const { launchOptions = {} } = launchContext;
42
48
  if (launchOptions.executablePath) {
43
49
  return launchOptions.executablePath;
@@ -79,12 +85,11 @@ function getDefaultExecutablePath(launchContext, config) {
79
85
  * Optional settings passed to `browserType.launch()`. In addition to
80
86
  * [Playwright's options](https://playwright.dev/docs/api/class-browsertype?_highlight=launch#browsertypelaunchoptions)
81
87
  * the object may contain our own {@link PlaywrightLaunchContext} that enable additional features.
82
- * @param [config]
88
+ * @param [configuration]
83
89
  * @returns
84
90
  * Promise that resolves to Playwright's `Browser` instance.
85
91
  */
86
- export async function launchPlaywright(launchContext, config = Configuration.getGlobalConfig()) {
87
- const playwrightLauncher = new PlaywrightLauncher(launchContext, config);
92
+ export async function launchPlaywright(launchContext, configuration = Configuration.getGlobalConfiguration()) {
93
+ const playwrightLauncher = new PlaywrightLauncher(launchContext, configuration);
88
94
  return playwrightLauncher.launch();
89
95
  }
90
- //# sourceMappingURL=playwright-launcher.js.map
@@ -17,14 +17,13 @@
17
17
  * ```
18
18
  * @module playwrightUtils
19
19
  */
20
- import { Configuration, type Request, type Session } from '@crawlee/browser';
21
- import type { BatchAddRequestsResult } from '@crawlee/types';
22
- import { type CheerioRoot, type Dictionary } from '@crawlee/utils';
20
+ import { Configuration, type Request } from '@crawlee/browser';
21
+ import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
22
+ import type { CheerioAPI } from 'cheerio';
23
23
  // @ts-ignore optional peer dependency or compatibility with es2022
24
- import type { Page, Response } from 'playwright';
24
+ import type { Download, Page, Response } from 'playwright';
25
25
  import type { EnqueueLinksByClickingElementsOptions } from '../enqueue-links/click-elements.js';
26
26
  import { enqueueLinksByClickingElements } from '../enqueue-links/click-elements.js';
27
- import { RenderingTypePredictor } from './rendering-type-prediction.js';
28
27
  export interface InjectFileOptions {
29
28
  /**
30
29
  * Enables the injected script to survive page navigations and reloads without need to be re-injected manually.
@@ -265,9 +264,9 @@ export interface SaveSnapshotOptions {
265
264
  keyValueStoreName?: string | null;
266
265
  /**
267
266
  * Configuration of the crawler that will be used to save the snapshot.
268
- * @default Configuration.getGlobalConfig()
267
+ * @default Configuration.getGlobalConfiguration()
269
268
  */
270
- config?: Configuration;
269
+ configuration?: Configuration;
271
270
  }
272
271
  /**
273
272
  * Saves a full screenshot and HTML of the current page into a Key-Value store.
@@ -287,8 +286,7 @@ export declare function saveSnapshot(page: Page, options?: SaveSnapshotOptions):
287
286
  * @param page Playwright [`Page`](https://playwright.dev/docs/api/class-page) object.
288
287
  * @param ignoreShadowRoots
289
288
  */
290
- export declare function parseWithCheerio(page: Page, ignoreShadowRoots?: boolean, ignoreIframes?: boolean): Promise<CheerioRoot>;
291
- export declare function closeCookieModals(page: Page): Promise<void>;
289
+ export declare function parseWithCheerio(page: Page, ignoreShadowRoots?: boolean, ignoreIframes?: boolean): Promise<CheerioAPI>;
292
290
  export interface HandleCloudflareChallengeOptions {
293
291
  /** Logging defaults to the `debug` level, use this flag to log to `info` level instead. */
294
292
  verbose?: boolean;
@@ -303,6 +301,13 @@ export interface HandleCloudflareChallengeOptions {
303
301
  isChallengeCallback?: (page: Page) => Promise<boolean>;
304
302
  /** Allows overriding the detection of Cloudflare "blocked page". */
305
303
  isBlockedCallback?: (page: Page) => Promise<boolean>;
304
+ /** Allows overriding how the checkbox click position is calculated. */
305
+ clickPositionCallback?: (page: Page) => Promise<{
306
+ x: number;
307
+ y: number;
308
+ } | null>;
309
+ /** Optional delay (in seconds) before the first click attempt on the challenge checkbox. Defaults to 1s. */
310
+ preChallengeSleepSecs?: number;
306
311
  }
307
312
  /**
308
313
  * This helper tries to solve the Cloudflare challenge automatically by clicking on the checkbox.
@@ -311,24 +316,24 @@ export interface HandleCloudflareChallengeOptions {
311
316
  * result in a SessionError which will be automatically retried, so only successful requests will get
312
317
  * into the `requestHandler`.
313
318
  *
319
+ * On a successfully solved challenge the page is reloaded and the new {@link Response} is returned, so
320
+ * it can be propagated back to the crawling context via a hook return value (see
321
+ * {@link handleCloudflareChallengeHook}).
322
+ *
314
323
  * Works best with camoufox.
315
324
  *
316
325
  * **Example usage**
317
326
  * ```ts
318
327
  * postNavigationHooks: [
319
- * async ({ handleCloudflareChallenge }) => {
320
- * await handleCloudflareChallenge();
321
- * },
328
+ * async (context) => ({ response: await context.handleCloudflareChallenge() }),
322
329
  * ],
323
330
  * ```
324
331
  *
325
332
  * @param page Playwright [`Page`](https://playwright.dev/docs/api/class-page) object
326
333
  * @param url current URL for request identification, only used for logging
327
- * @param [session] current session object
328
334
  * @param [options]
329
335
  */
330
- declare function handleCloudflareChallenge(page: Page, url: string, session?: Session, options?: HandleCloudflareChallengeOptions): Promise<void>;
331
- /** @internal */
336
+ export declare function handleCloudflareChallenge(page: Page, url: string, options?: HandleCloudflareChallengeOptions): Promise<Response | undefined>;
332
337
  export interface PlaywrightContextUtils {
333
338
  /**
334
339
  * Injects a JavaScript file into current `page`.
@@ -427,7 +432,7 @@ export interface PlaywrightContextUtils {
427
432
  * });
428
433
  * ```
429
434
  */
430
- parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioRoot>;
435
+ parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
431
436
  /**
432
437
  * Scrolls to the bottom of a page, or until it times out.
433
438
  * Loads dynamic content when it hits the bottom of a page, and then continues scrolling.
@@ -447,8 +452,7 @@ export interface PlaywrightContextUtils {
447
452
  * in `href` elements, but rather navigations are triggered in click handlers.
448
453
  * If you're looking to find URLs in `href` attributes of the page, see {@link enqueueLinks}.
449
454
  *
450
- * Optionally, the function allows you to filter the target links' URLs using an array of {@link PseudoUrl} objects
451
- * and override settings of the enqueued {@link Request} objects.
455
+ * Optionally, the function allows you to filter the target links' URLs using an array of glob or regexp patterns.
452
456
  *
453
457
  * **IMPORTANT**: To be able to do this, this function uses various mutations on the page,
454
458
  * such as changing the Z-index of elements being clicked and their visibility. Therefore,
@@ -469,9 +473,9 @@ export interface PlaywrightContextUtils {
469
473
  * async requestHandler({ enqueueLinksByClickingElements }) {
470
474
  * await enqueueLinksByClickingElements({
471
475
  * selector: 'a.product-detail',
472
- * globs: [
473
- * 'https://www.example.com/handbags/**'
474
- * 'https://www.example.com/purses/**'
476
+ * include: [
477
+ * 'https://www.example.com/handbags/**',
478
+ * 'https://www.example.com/purses/**',
475
479
  * ],
476
480
  * });
477
481
  * });
@@ -479,7 +483,7 @@ export interface PlaywrightContextUtils {
479
483
  *
480
484
  * @returns Promise that resolves to {@link BatchAddRequestsResult} object.
481
485
  */
482
- enqueueLinksByClickingElements(options: Omit<EnqueueLinksByClickingElementsOptions, 'page' | 'requestQueue'>): Promise<BatchAddRequestsResult>;
486
+ enqueueLinksByClickingElements(options: Omit<EnqueueLinksByClickingElementsOptions, 'page' | 'requestManager'>): Promise<BatchAddRequestsResult>;
483
487
  /**
484
488
  * Compiles a Playwright script into an async function that may be executed at any time
485
489
  * by providing it with the following object:
@@ -507,19 +511,6 @@ export interface PlaywrightContextUtils {
507
511
  * secured copies beforehand.
508
512
  */
509
513
  compileScript(scriptString: string, ctx?: Dictionary): CompiledScriptFunction;
510
- /**
511
- * Tries to close cookie consent modals on the page. Based on the I Don't Care About Cookies browser extension.
512
- *
513
- * Note that this method requires the idcac-playwright package to be installed.
514
- * Crawlee does not include it by default due to licensing issues.
515
- *
516
- * To use this method, please install the package manually by running:
517
- *
518
- * ```bash
519
- * npm install idcac-playwright
520
- * ```
521
- */
522
- closeCookieModals(): Promise<void>;
523
514
  /**
524
515
  * This helper tries to solve the Cloudflare challenge automatically by clicking on the checkbox.
525
516
  * It will try to detect the Cloudflare page, click on the checkbox, and wait for 10 seconds (configurable
@@ -527,35 +518,42 @@ export interface PlaywrightContextUtils {
527
518
  * result in a SessionError which will be automatically retried, so only successful requests will get
528
519
  * into the `requestHandler`.
529
520
  *
530
- * Works best with camoufox.
521
+ * On a successfully solved challenge the page is reloaded and the new {@link Response} is returned,
522
+ * which can be returned from the hook to update the crawling context's `response`. For the common case,
523
+ * prefer the pre-wrapped {@link handleCloudflareChallengeHook} hook.
531
524
  *
532
525
  * **Example usage**
533
526
  * ```ts
534
527
  * postNavigationHooks: [
535
- * async ({ handleCloudflareChallenge }) => {
536
- * await handleCloudflareChallenge();
537
- * },
528
+ * async (context) => ({ response: await context.handleCloudflareChallenge() }),
538
529
  * ],
539
530
  * ```
540
531
  *
541
532
  * @param [options]
542
533
  */
543
- handleCloudflareChallenge(options?: HandleCloudflareChallengeOptions): Promise<void>;
534
+ handleCloudflareChallenge(options?: HandleCloudflareChallengeOptions): Promise<Response | undefined>;
535
+ /**
536
+ * Returns the list of {@link https://playwright.dev/docs/api/class-download | Download} objects
537
+ * collected during the current page navigation and request handler.
538
+ *
539
+ * Useful for accessing files that the page downloads automatically.
540
+ * For most use cases, prefer re-enqueueing the URL to {@link FileDownload}.
541
+ * Use this only when direct access to the Playwright `Download` object is required.
542
+ *
543
+ * **Example usage**
544
+ * ```ts
545
+ * requestHandler: async ({ listDownloads }) => {
546
+ * for (const download of await listDownloads()) {
547
+ * try {
548
+ * const stream = await download.createReadStream();
549
+ * // stream to storage...
550
+ * } catch {
551
+ * // download failed or was cancelled
552
+ * }
553
+ * }
554
+ * },
555
+ * ```
556
+ */
557
+ listDownloads(): Promise<Download[]>;
544
558
  }
545
559
  export { enqueueLinksByClickingElements };
546
- /** @internal */
547
- export declare const playwrightUtils: {
548
- injectFile: typeof injectFile;
549
- injectJQuery: typeof injectJQuery;
550
- gotoExtended: typeof gotoExtended;
551
- blockRequests: typeof blockRequests;
552
- enqueueLinksByClickingElements: typeof enqueueLinksByClickingElements;
553
- parseWithCheerio: typeof parseWithCheerio;
554
- infiniteScroll: typeof infiniteScroll;
555
- saveSnapshot: typeof saveSnapshot;
556
- compileScript: typeof compileScript;
557
- closeCookieModals: typeof closeCookieModals;
558
- RenderingTypePredictor: typeof RenderingTypePredictor;
559
- handleCloudflareChallenge: typeof handleCloudflareChallenge;
560
- };
561
- //# sourceMappingURL=playwright-utils.d.ts.map