@crawlee/playwright 4.0.0-beta.2 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +4 -3
- package/index.js +2 -2
- package/internals/adaptive-playwright-crawler.d.ts +146 -91
- package/internals/adaptive-playwright-crawler.js +485 -268
- package/internals/enqueue-links/click-elements.d.ts +37 -55
- package/internals/enqueue-links/click-elements.js +65 -55
- package/internals/playwright-browser-pool.d.ts +71 -0
- package/internals/playwright-browser-pool.js +61 -0
- package/internals/playwright-crawler.d.ts +177 -172
- package/internals/playwright-crawler.js +103 -61
- package/internals/playwright-launcher.d.ts +30 -20
- package/internals/playwright-launcher.js +22 -17
- package/internals/utils/playwright-utils.d.ts +55 -50
- package/internals/utils/playwright-utils.js +121 -148
- package/internals/utils/rendering-type-prediction.d.ts +44 -13
- package/internals/utils/rendering-type-prediction.js +95 -29
- package/package.json +15 -15
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
- package/internals/adaptive-playwright-crawler.js.map +0 -1
- package/internals/enqueue-links/click-elements.d.ts.map +0 -1
- package/internals/enqueue-links/click-elements.js.map +0 -1
- package/internals/playwright-crawler.d.ts.map +0 -1
- package/internals/playwright-crawler.js.map +0 -1
- package/internals/playwright-launcher.d.ts.map +0 -1
- package/internals/playwright-launcher.js.map +0 -1
- package/internals/utils/playwright-utils.d.ts.map +0 -1
- package/internals/utils/playwright-utils.js.map +0 -1
- package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
- package/internals/utils/rendering-type-prediction.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
import { BrowserCrawler,
|
|
2
|
-
import
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
1
|
+
import { BrowserCrawler, RequestState, Router, serviceLocator } from '@crawlee/browser';
|
|
2
|
+
import { assertBrowserPoolNotConfigured, parseArgument, schemas } from '@crawlee/utils/internal';
|
|
3
|
+
import { z } from 'zod';
|
|
4
|
+
import { playwrightBrowserPool, remotePlaywrightBrowserPool } from './playwright-browser-pool.js';
|
|
5
|
+
import { blockRequests, compileScript, enqueueLinksByClickingElements, gotoExtended, handleCloudflareChallenge, infiniteScroll, injectFile, injectJQuery, parseWithCheerio, saveSnapshot, } from './utils/playwright-utils.js';
|
|
5
6
|
/**
|
|
6
7
|
* Provides a simple framework for parallel crawling of web pages
|
|
7
8
|
* using headless Chromium, Firefox and Webkit browsers with [Playwright](https://github.com/microsoft/playwright).
|
|
@@ -13,24 +14,26 @@ import { gotoExtended, registerUtilsToContext } from './utils/playwright-utils.j
|
|
|
13
14
|
* If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
|
|
14
15
|
* which downloads the pages using raw HTTP requests and is about 10x faster.
|
|
15
16
|
*
|
|
16
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
17
|
-
* {@link
|
|
18
|
-
*
|
|
17
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
18
|
+
* {@link IRequestManager|request manager} provided via the {@link PlaywrightCrawlerOptions.requestManager|`requestManager`}
|
|
19
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
20
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
21
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
22
|
+
* result as `requestManager`.
|
|
19
23
|
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
24
|
+
* > The {@link PlaywrightCrawlerOptions.requestList|`requestList`} and {@link PlaywrightCrawlerOptions.requestQueue|`requestQueue`}
|
|
25
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
23
26
|
*
|
|
24
27
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
25
28
|
*
|
|
26
29
|
* `PlaywrightCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
|
|
27
30
|
* and then calls the function provided by user as the {@link PlaywrightCrawlerOptions.requestHandler} option.
|
|
28
31
|
*
|
|
29
|
-
* New pages are only opened when there is enough free CPU and memory available,
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
* {@link
|
|
32
|
+
* New pages are only opened when there is enough free CPU and memory available, as judged by the crawler's
|
|
33
|
+
* {@link ConcurrencySystem}.
|
|
34
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
35
|
+
* `PlaywrightCrawler` constructor, or, for finer control, by injecting a pre-configured
|
|
36
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
34
37
|
*
|
|
35
38
|
* Note that the pool of Playwright instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
|
|
36
39
|
*
|
|
@@ -66,74 +69,113 @@ import { gotoExtended, registerUtilsToContext } from './utils/playwright-utils.j
|
|
|
66
69
|
* @category Crawlers
|
|
67
70
|
*/
|
|
68
71
|
export class PlaywrightCrawler extends BrowserCrawler {
|
|
69
|
-
|
|
70
|
-
|
|
72
|
+
/**
|
|
73
|
+
* @internal
|
|
74
|
+
*/
|
|
71
75
|
static optionsShape = {
|
|
72
76
|
...BrowserCrawler.optionsShape,
|
|
73
|
-
|
|
74
|
-
|
|
77
|
+
launchContext: schemas.anyObject.default(() => ({})),
|
|
78
|
+
headless: z.boolean().optional(),
|
|
75
79
|
};
|
|
80
|
+
/** @internal */
|
|
81
|
+
static optionsSchema = z.strictObject(PlaywrightCrawler.optionsShape);
|
|
76
82
|
/**
|
|
77
83
|
* All `PlaywrightCrawler` parameters are passed via an options object.
|
|
78
84
|
*/
|
|
79
|
-
constructor(options = {}
|
|
80
|
-
|
|
81
|
-
const { launchContext
|
|
82
|
-
const browserPoolOptions = {
|
|
83
|
-
...options.browserPoolOptions,
|
|
84
|
-
};
|
|
85
|
+
constructor(options = {}) {
|
|
86
|
+
const parsedOptions = parseArgument(options, PlaywrightCrawler.optionsSchema, 'PlaywrightCrawlerOptions');
|
|
87
|
+
const { launchContext, headless, configuration, ...browserCrawlerOptions } = parsedOptions;
|
|
85
88
|
if (launchContext.proxyUrl) {
|
|
86
89
|
throw new Error('PlaywrightCrawlerOptions.launchContext.proxyUrl is not allowed in PlaywrightCrawler.' +
|
|
87
90
|
'Use PlaywrightCrawlerOptions.proxyConfiguration');
|
|
88
91
|
}
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
launchContext.launchOptions ??= {};
|
|
96
|
-
launchContext.launchOptions.headless = headless;
|
|
92
|
+
if (options.browserPool) {
|
|
93
|
+
// The raw options, not the parsed ones: `launchContext` has a default, so by now it is always set.
|
|
94
|
+
assertBrowserPoolNotConfigured(new.target.name, {
|
|
95
|
+
launchContext: options.launchContext,
|
|
96
|
+
headless: options.headless,
|
|
97
|
+
});
|
|
97
98
|
}
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
99
|
+
super({
|
|
100
|
+
...browserCrawlerOptions,
|
|
101
|
+
configuration,
|
|
102
|
+
browserPoolBuilder: (remoteBrowser) => remoteBrowser
|
|
103
|
+
? remotePlaywrightBrowserPool({ ...remoteBrowser, launchContext, headless, configuration })
|
|
104
|
+
: playwrightBrowserPool({ launchContext, headless, configuration }),
|
|
105
|
+
contextPipelineBuilder: () => this.#buildContextPipeline(),
|
|
106
|
+
});
|
|
103
107
|
}
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
await super._runRequestHandler(context);
|
|
108
|
+
#buildContextPipeline() {
|
|
109
|
+
return this.buildContextPipeline().compose(this.enhanceContext.bind(this));
|
|
107
110
|
}
|
|
108
|
-
async
|
|
111
|
+
async navigationHandler(crawlingContext, gotoOptions) {
|
|
109
112
|
return gotoExtended(crawlingContext.page, crawlingContext.request, gotoOptions);
|
|
110
113
|
}
|
|
114
|
+
async enhanceContext(context) {
|
|
115
|
+
const waitForSelector = async (selector, timeoutMs = 5_000) => {
|
|
116
|
+
const locator = context.page.locator(selector).first();
|
|
117
|
+
await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
|
|
118
|
+
};
|
|
119
|
+
const downloads = [];
|
|
120
|
+
context.page.on('download', (download) => downloads.push(download));
|
|
121
|
+
return {
|
|
122
|
+
injectFile: async (filePath, options) => injectFile(context.page, filePath, options),
|
|
123
|
+
injectJQuery: async () => {
|
|
124
|
+
if (context.request.state === RequestState.BEFORE_NAV) {
|
|
125
|
+
context.log.warning('Using injectJQuery() in preNavigationHooks leads to unstable results. Use it in a postNavigationHook or a requestHandler instead.');
|
|
126
|
+
await injectJQuery(context.page);
|
|
127
|
+
return;
|
|
128
|
+
}
|
|
129
|
+
await injectJQuery(context.page, { surviveNavigations: false });
|
|
130
|
+
},
|
|
131
|
+
blockRequests: async (options) => blockRequests(context.page, options),
|
|
132
|
+
waitForSelector,
|
|
133
|
+
parseWithCheerio: async (selector, timeoutMs = 5_000) => {
|
|
134
|
+
if (selector) {
|
|
135
|
+
await waitForSelector(selector, timeoutMs);
|
|
136
|
+
}
|
|
137
|
+
return parseWithCheerio(context.page, this.ignoreShadowRoots, this.ignoreIframes);
|
|
138
|
+
},
|
|
139
|
+
infiniteScroll: async (options) => infiniteScroll(context.page, options),
|
|
140
|
+
listDownloads: async () => downloads,
|
|
141
|
+
saveSnapshot: async (options) => saveSnapshot(context.page, {
|
|
142
|
+
...options,
|
|
143
|
+
configuration: serviceLocator.getConfiguration(),
|
|
144
|
+
}),
|
|
145
|
+
enqueueLinksByClickingElements: async (options) => enqueueLinksByClickingElements({
|
|
146
|
+
...options,
|
|
147
|
+
page: context.page,
|
|
148
|
+
requestManager: this.requestManager,
|
|
149
|
+
}),
|
|
150
|
+
compileScript: (scriptString, ctx) => compileScript(scriptString, ctx),
|
|
151
|
+
handleCloudflareChallenge: async (options) => {
|
|
152
|
+
return handleCloudflareChallenge(context.page, context.request.url, options);
|
|
153
|
+
},
|
|
154
|
+
};
|
|
155
|
+
}
|
|
111
156
|
}
|
|
112
157
|
/**
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
* Defaults to the {@link PlaywrightCrawlingContext}.
|
|
116
|
-
*
|
|
117
|
-
* > Serves as a shortcut for using `Router.create<PlaywrightCrawlingContext>()`.
|
|
158
|
+
* Returns a `postNavigationHooks`-ready hook that runs {@link PlaywrightContextUtils.handleCloudflareChallenge}
|
|
159
|
+
* and propagates the post-challenge {@link Response} back into the crawling context via its return value.
|
|
118
160
|
*
|
|
161
|
+
* **Example usage**
|
|
119
162
|
* ```ts
|
|
120
|
-
* import { PlaywrightCrawler,
|
|
121
|
-
*
|
|
122
|
-
* const router = createPlaywrightRouter();
|
|
123
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
124
|
-
* ctx.log.info('...');
|
|
125
|
-
* });
|
|
126
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
127
|
-
* ctx.log.info('...');
|
|
128
|
-
* });
|
|
163
|
+
* import { PlaywrightCrawler, handleCloudflareChallengeHook } from 'crawlee';
|
|
129
164
|
*
|
|
130
165
|
* const crawler = new PlaywrightCrawler({
|
|
131
|
-
*
|
|
166
|
+
* postNavigationHooks: [handleCloudflareChallengeHook()],
|
|
132
167
|
* });
|
|
133
|
-
* await crawler.run();
|
|
134
168
|
* ```
|
|
135
169
|
*/
|
|
136
|
-
export function
|
|
137
|
-
return
|
|
170
|
+
export function handleCloudflareChallengeHook(options) {
|
|
171
|
+
return async (context) => {
|
|
172
|
+
const response = await context.handleCloudflareChallenge(options);
|
|
173
|
+
if (response !== undefined) {
|
|
174
|
+
return { response };
|
|
175
|
+
}
|
|
176
|
+
return undefined;
|
|
177
|
+
};
|
|
178
|
+
}
|
|
179
|
+
export function createPlaywrightRouter(routesOrSchemas) {
|
|
180
|
+
return Router.create(routesOrSchemas);
|
|
138
181
|
}
|
|
139
|
-
//# sourceMappingURL=playwright-crawler.js.map
|
|
@@ -3,6 +3,7 @@ import { BrowserLauncher, Configuration } from '@crawlee/browser';
|
|
|
3
3
|
import { PlaywrightPlugin } from '@crawlee/browser-pool';
|
|
4
4
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
5
5
|
import type { Browser, BrowserType, LaunchOptions } from 'playwright';
|
|
6
|
+
import { z } from 'zod';
|
|
6
7
|
/**
|
|
7
8
|
* Apify extends the launch options of Playwright.
|
|
8
9
|
* You can use any of the Playwright compatible
|
|
@@ -70,31 +71,41 @@ export interface PlaywrightLaunchContext extends BrowserLaunchContext<LaunchOpti
|
|
|
70
71
|
* @ignore
|
|
71
72
|
*/
|
|
72
73
|
export declare class PlaywrightLauncher extends BrowserLauncher<PlaywrightPlugin> {
|
|
73
|
-
readonly
|
|
74
|
+
readonly configuration: Configuration;
|
|
75
|
+
/**
|
|
76
|
+
* @internal
|
|
77
|
+
*/
|
|
74
78
|
protected static optionsShape: {
|
|
79
|
+
proxyUrl: z.ZodOptional<z.ZodURL>;
|
|
80
|
+
useChrome: z.ZodOptional<z.ZodBoolean>;
|
|
81
|
+
useIncognitoPages: z.ZodOptional<z.ZodBoolean>;
|
|
82
|
+
browserPerProxy: z.ZodOptional<z.ZodBoolean>;
|
|
83
|
+
ignoreProxyCertificate: z.ZodOptional<z.ZodBoolean>;
|
|
84
|
+
userDataDir: z.ZodOptional<z.ZodString>;
|
|
75
85
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
launchContextOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
79
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
80
|
-
proxyUrl: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
|
|
81
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
82
|
-
useChrome: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
86
|
+
launchOptions: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
|
|
87
|
+
userAgent: z.ZodOptional<z.ZodString>;
|
|
83
88
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
+
launcher: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
|
|
90
|
+
};
|
|
91
|
+
/** @internal */
|
|
92
|
+
protected static optionsSchema: z.ZodObject<{
|
|
93
|
+
proxyUrl: z.ZodOptional<z.ZodURL>;
|
|
94
|
+
useChrome: z.ZodOptional<z.ZodBoolean>;
|
|
95
|
+
useIncognitoPages: z.ZodOptional<z.ZodBoolean>;
|
|
96
|
+
browserPerProxy: z.ZodOptional<z.ZodBoolean>;
|
|
97
|
+
ignoreProxyCertificate: z.ZodOptional<z.ZodBoolean>;
|
|
98
|
+
userDataDir: z.ZodOptional<z.ZodString>;
|
|
89
99
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
90
|
-
launchOptions: import("
|
|
100
|
+
launchOptions: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
|
|
101
|
+
userAgent: z.ZodOptional<z.ZodString>;
|
|
91
102
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
92
|
-
|
|
93
|
-
}
|
|
103
|
+
launcher: z.ZodOptional<z.ZodCustom<import("@crawlee/types").Dictionary, import("@crawlee/types").Dictionary>>;
|
|
104
|
+
}, z.core.$strict>;
|
|
94
105
|
/**
|
|
95
106
|
* All `PlaywrightLauncher` parameters are passed via this launchContext object.
|
|
96
107
|
*/
|
|
97
|
-
constructor(launchContext?: PlaywrightLaunchContext,
|
|
108
|
+
constructor(launchContext?: PlaywrightLaunchContext, configuration?: Configuration);
|
|
98
109
|
}
|
|
99
110
|
/**
|
|
100
111
|
* Launches headless browsers using Playwright pre-configured to work within the Apify platform.
|
|
@@ -125,9 +136,8 @@ export declare class PlaywrightLauncher extends BrowserLauncher<PlaywrightPlugin
|
|
|
125
136
|
* Optional settings passed to `browserType.launch()`. In addition to
|
|
126
137
|
* [Playwright's options](https://playwright.dev/docs/api/class-browsertype?_highlight=launch#browsertypelaunchoptions)
|
|
127
138
|
* the object may contain our own {@link PlaywrightLaunchContext} that enable additional features.
|
|
128
|
-
* @param [
|
|
139
|
+
* @param [configuration]
|
|
129
140
|
* @returns
|
|
130
141
|
* Promise that resolves to Playwright's `Browser` instance.
|
|
131
142
|
*/
|
|
132
|
-
export declare function launchPlaywright(launchContext?: PlaywrightLaunchContext,
|
|
133
|
-
//# sourceMappingURL=playwright-launcher.d.ts.map
|
|
143
|
+
export declare function launchPlaywright(launchContext?: PlaywrightLaunchContext, configuration?: Configuration): Promise<Browser>;
|
|
@@ -1,33 +1,39 @@
|
|
|
1
1
|
import { BrowserLauncher, Configuration } from '@crawlee/browser';
|
|
2
2
|
import { PlaywrightPlugin } from '@crawlee/browser-pool';
|
|
3
|
-
import
|
|
3
|
+
import { parseArgument, schemas } from '@crawlee/utils/internal';
|
|
4
|
+
import { z } from 'zod';
|
|
4
5
|
/**
|
|
5
6
|
* `PlaywrightLauncher` is based on the `BrowserLauncher`. It launches `playwright` browser instance.
|
|
6
7
|
* @ignore
|
|
7
8
|
*/
|
|
8
9
|
export class PlaywrightLauncher extends BrowserLauncher {
|
|
9
|
-
|
|
10
|
+
configuration;
|
|
11
|
+
/**
|
|
12
|
+
* @internal
|
|
13
|
+
*/
|
|
10
14
|
static optionsShape = {
|
|
11
15
|
...BrowserLauncher.optionsShape,
|
|
12
|
-
launcher
|
|
13
|
-
|
|
16
|
+
// Passthrough schema — the launcher module object must keep its prototype through parsing.
|
|
17
|
+
launcher: schemas.anyObject.optional(),
|
|
14
18
|
};
|
|
19
|
+
/** @internal */
|
|
20
|
+
static optionsSchema = z.strictObject(PlaywrightLauncher.optionsShape);
|
|
15
21
|
/**
|
|
16
22
|
* All `PlaywrightLauncher` parameters are passed via this launchContext object.
|
|
17
23
|
*/
|
|
18
|
-
constructor(launchContext = {},
|
|
19
|
-
|
|
20
|
-
const { launcher = BrowserLauncher.requireLauncherOrThrow('playwright', 'apify/actor-node-playwright-*').chromium, } =
|
|
21
|
-
const { launchOptions = {}, ...rest } =
|
|
24
|
+
constructor(launchContext = {}, configuration = Configuration.getGlobalConfiguration()) {
|
|
25
|
+
const parsedContext = parseArgument(launchContext, PlaywrightLauncher.optionsSchema, 'PlaywrightLaunchContext');
|
|
26
|
+
const { launcher = BrowserLauncher.requireLauncherOrThrow('playwright', 'apify/actor-node-playwright-*').chromium, } = parsedContext;
|
|
27
|
+
const { launchOptions = {}, ...rest } = parsedContext;
|
|
22
28
|
super({
|
|
23
29
|
...rest,
|
|
24
30
|
launchOptions: {
|
|
25
31
|
...launchOptions,
|
|
26
|
-
executablePath: getDefaultExecutablePath(
|
|
32
|
+
executablePath: getDefaultExecutablePath(parsedContext, configuration),
|
|
27
33
|
},
|
|
28
34
|
launcher,
|
|
29
|
-
},
|
|
30
|
-
this.
|
|
35
|
+
}, configuration);
|
|
36
|
+
this.configuration = configuration;
|
|
31
37
|
this.Plugin = PlaywrightPlugin;
|
|
32
38
|
}
|
|
33
39
|
}
|
|
@@ -36,8 +42,8 @@ export class PlaywrightLauncher extends BrowserLauncher {
|
|
|
36
42
|
* @returns default path to browser.
|
|
37
43
|
* @ignore
|
|
38
44
|
*/
|
|
39
|
-
function getDefaultExecutablePath(launchContext,
|
|
40
|
-
const pathFromPlaywrightImage =
|
|
45
|
+
function getDefaultExecutablePath(launchContext, configuration) {
|
|
46
|
+
const pathFromPlaywrightImage = configuration.defaultBrowserPath;
|
|
41
47
|
const { launchOptions = {} } = launchContext;
|
|
42
48
|
if (launchOptions.executablePath) {
|
|
43
49
|
return launchOptions.executablePath;
|
|
@@ -79,12 +85,11 @@ function getDefaultExecutablePath(launchContext, config) {
|
|
|
79
85
|
* Optional settings passed to `browserType.launch()`. In addition to
|
|
80
86
|
* [Playwright's options](https://playwright.dev/docs/api/class-browsertype?_highlight=launch#browsertypelaunchoptions)
|
|
81
87
|
* the object may contain our own {@link PlaywrightLaunchContext} that enable additional features.
|
|
82
|
-
* @param [
|
|
88
|
+
* @param [configuration]
|
|
83
89
|
* @returns
|
|
84
90
|
* Promise that resolves to Playwright's `Browser` instance.
|
|
85
91
|
*/
|
|
86
|
-
export async function launchPlaywright(launchContext,
|
|
87
|
-
const playwrightLauncher = new PlaywrightLauncher(launchContext,
|
|
92
|
+
export async function launchPlaywright(launchContext, configuration = Configuration.getGlobalConfiguration()) {
|
|
93
|
+
const playwrightLauncher = new PlaywrightLauncher(launchContext, configuration);
|
|
88
94
|
return playwrightLauncher.launch();
|
|
89
95
|
}
|
|
90
|
-
//# sourceMappingURL=playwright-launcher.js.map
|
|
@@ -17,15 +17,13 @@
|
|
|
17
17
|
* ```
|
|
18
18
|
* @module playwrightUtils
|
|
19
19
|
*/
|
|
20
|
-
import { Configuration, type Request
|
|
21
|
-
import type { BatchAddRequestsResult } from '@crawlee/types';
|
|
22
|
-
import
|
|
20
|
+
import { Configuration, type Request } from '@crawlee/browser';
|
|
21
|
+
import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
|
|
22
|
+
import type { CheerioAPI } from 'cheerio';
|
|
23
23
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
24
|
-
import type { Page, Response } from 'playwright';
|
|
24
|
+
import type { Download, Page, Response } from 'playwright';
|
|
25
25
|
import type { EnqueueLinksByClickingElementsOptions } from '../enqueue-links/click-elements.js';
|
|
26
26
|
import { enqueueLinksByClickingElements } from '../enqueue-links/click-elements.js';
|
|
27
|
-
import type { PlaywrightCrawlerOptions, PlaywrightCrawlingContext } from '../playwright-crawler.js';
|
|
28
|
-
import { RenderingTypePredictor } from './rendering-type-prediction.js';
|
|
29
27
|
export interface InjectFileOptions {
|
|
30
28
|
/**
|
|
31
29
|
* Enables the injected script to survive page navigations and reloads without need to be re-injected manually.
|
|
@@ -266,9 +264,9 @@ export interface SaveSnapshotOptions {
|
|
|
266
264
|
keyValueStoreName?: string | null;
|
|
267
265
|
/**
|
|
268
266
|
* Configuration of the crawler that will be used to save the snapshot.
|
|
269
|
-
* @default Configuration.
|
|
267
|
+
* @default Configuration.getGlobalConfiguration()
|
|
270
268
|
*/
|
|
271
|
-
|
|
269
|
+
configuration?: Configuration;
|
|
272
270
|
}
|
|
273
271
|
/**
|
|
274
272
|
* Saves a full screenshot and HTML of the current page into a Key-Value store.
|
|
@@ -288,9 +286,8 @@ export declare function saveSnapshot(page: Page, options?: SaveSnapshotOptions):
|
|
|
288
286
|
* @param page Playwright [`Page`](https://playwright.dev/docs/api/class-page) object.
|
|
289
287
|
* @param ignoreShadowRoots
|
|
290
288
|
*/
|
|
291
|
-
export declare function parseWithCheerio(page: Page, ignoreShadowRoots?: boolean, ignoreIframes?: boolean): Promise<
|
|
292
|
-
export
|
|
293
|
-
interface HandleCloudflareChallengeOptions {
|
|
289
|
+
export declare function parseWithCheerio(page: Page, ignoreShadowRoots?: boolean, ignoreIframes?: boolean): Promise<CheerioAPI>;
|
|
290
|
+
export interface HandleCloudflareChallengeOptions {
|
|
294
291
|
/** Logging defaults to the `debug` level, use this flag to log to `info` level instead. */
|
|
295
292
|
verbose?: boolean;
|
|
296
293
|
/** How long should we wait after the challenge is completed for the final page to load. */
|
|
@@ -304,6 +301,13 @@ interface HandleCloudflareChallengeOptions {
|
|
|
304
301
|
isChallengeCallback?: (page: Page) => Promise<boolean>;
|
|
305
302
|
/** Allows overriding the detection of Cloudflare "blocked page". */
|
|
306
303
|
isBlockedCallback?: (page: Page) => Promise<boolean>;
|
|
304
|
+
/** Allows overriding how the checkbox click position is calculated. */
|
|
305
|
+
clickPositionCallback?: (page: Page) => Promise<{
|
|
306
|
+
x: number;
|
|
307
|
+
y: number;
|
|
308
|
+
} | null>;
|
|
309
|
+
/** Optional delay (in seconds) before the first click attempt on the challenge checkbox. Defaults to 1s. */
|
|
310
|
+
preChallengeSleepSecs?: number;
|
|
307
311
|
}
|
|
308
312
|
/**
|
|
309
313
|
* This helper tries to solve the Cloudflare challenge automatically by clicking on the checkbox.
|
|
@@ -312,24 +316,24 @@ interface HandleCloudflareChallengeOptions {
|
|
|
312
316
|
* result in a SessionError which will be automatically retried, so only successful requests will get
|
|
313
317
|
* into the `requestHandler`.
|
|
314
318
|
*
|
|
319
|
+
* On a successfully solved challenge the page is reloaded and the new {@link Response} is returned, so
|
|
320
|
+
* it can be propagated back to the crawling context via a hook return value (see
|
|
321
|
+
* {@link handleCloudflareChallengeHook}).
|
|
322
|
+
*
|
|
315
323
|
* Works best with camoufox.
|
|
316
324
|
*
|
|
317
325
|
* **Example usage**
|
|
318
326
|
* ```ts
|
|
319
327
|
* postNavigationHooks: [
|
|
320
|
-
* async ({ handleCloudflareChallenge })
|
|
321
|
-
* await handleCloudflareChallenge();
|
|
322
|
-
* },
|
|
328
|
+
* async (context) => ({ response: await context.handleCloudflareChallenge() }),
|
|
323
329
|
* ],
|
|
324
330
|
* ```
|
|
325
331
|
*
|
|
326
332
|
* @param page Playwright [`Page`](https://playwright.dev/docs/api/class-page) object
|
|
327
333
|
* @param url current URL for request identification, only used for logging
|
|
328
|
-
* @param [session] current session object
|
|
329
334
|
* @param [options]
|
|
330
335
|
*/
|
|
331
|
-
declare function handleCloudflareChallenge(page: Page, url: string,
|
|
332
|
-
/** @internal */
|
|
336
|
+
export declare function handleCloudflareChallenge(page: Page, url: string, options?: HandleCloudflareChallengeOptions): Promise<Response | undefined>;
|
|
333
337
|
export interface PlaywrightContextUtils {
|
|
334
338
|
/**
|
|
335
339
|
* Injects a JavaScript file into current `page`.
|
|
@@ -428,7 +432,7 @@ export interface PlaywrightContextUtils {
|
|
|
428
432
|
* });
|
|
429
433
|
* ```
|
|
430
434
|
*/
|
|
431
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<
|
|
435
|
+
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
432
436
|
/**
|
|
433
437
|
* Scrolls to the bottom of a page, or until it times out.
|
|
434
438
|
* Loads dynamic content when it hits the bottom of a page, and then continues scrolling.
|
|
@@ -448,8 +452,7 @@ export interface PlaywrightContextUtils {
|
|
|
448
452
|
* in `href` elements, but rather navigations are triggered in click handlers.
|
|
449
453
|
* If you're looking to find URLs in `href` attributes of the page, see {@link enqueueLinks}.
|
|
450
454
|
*
|
|
451
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of
|
|
452
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
455
|
+
* Optionally, the function allows you to filter the target links' URLs using an array of glob or regexp patterns.
|
|
453
456
|
*
|
|
454
457
|
* **IMPORTANT**: To be able to do this, this function uses various mutations on the page,
|
|
455
458
|
* such as changing the Z-index of elements being clicked and their visibility. Therefore,
|
|
@@ -470,9 +473,9 @@ export interface PlaywrightContextUtils {
|
|
|
470
473
|
* async requestHandler({ enqueueLinksByClickingElements }) {
|
|
471
474
|
* await enqueueLinksByClickingElements({
|
|
472
475
|
* selector: 'a.product-detail',
|
|
473
|
-
*
|
|
474
|
-
* 'https://www.example.com/handbags/**'
|
|
475
|
-
* 'https://www.example.com/purses/**'
|
|
476
|
+
* include: [
|
|
477
|
+
* 'https://www.example.com/handbags/**',
|
|
478
|
+
* 'https://www.example.com/purses/**',
|
|
476
479
|
* ],
|
|
477
480
|
* });
|
|
478
481
|
* });
|
|
@@ -480,7 +483,7 @@ export interface PlaywrightContextUtils {
|
|
|
480
483
|
*
|
|
481
484
|
* @returns Promise that resolves to {@link BatchAddRequestsResult} object.
|
|
482
485
|
*/
|
|
483
|
-
enqueueLinksByClickingElements(options: Omit<EnqueueLinksByClickingElementsOptions, 'page' | '
|
|
486
|
+
enqueueLinksByClickingElements(options: Omit<EnqueueLinksByClickingElementsOptions, 'page' | 'requestManager'>): Promise<BatchAddRequestsResult>;
|
|
484
487
|
/**
|
|
485
488
|
* Compiles a Playwright script into an async function that may be executed at any time
|
|
486
489
|
* by providing it with the following object:
|
|
@@ -508,10 +511,6 @@ export interface PlaywrightContextUtils {
|
|
|
508
511
|
* secured copies beforehand.
|
|
509
512
|
*/
|
|
510
513
|
compileScript(scriptString: string, ctx?: Dictionary): CompiledScriptFunction;
|
|
511
|
-
/**
|
|
512
|
-
* Tries to close cookie consent modals on the page. Based on the I Don't Care About Cookies browser extension.
|
|
513
|
-
*/
|
|
514
|
-
closeCookieModals(): Promise<void>;
|
|
515
514
|
/**
|
|
516
515
|
* This helper tries to solve the Cloudflare challenge automatically by clicking on the checkbox.
|
|
517
516
|
* It will try to detect the Cloudflare page, click on the checkbox, and wait for 10 seconds (configurable
|
|
@@ -519,36 +518,42 @@ export interface PlaywrightContextUtils {
|
|
|
519
518
|
* result in a SessionError which will be automatically retried, so only successful requests will get
|
|
520
519
|
* into the `requestHandler`.
|
|
521
520
|
*
|
|
522
|
-
*
|
|
521
|
+
* On a successfully solved challenge the page is reloaded and the new {@link Response} is returned,
|
|
522
|
+
* which can be returned from the hook to update the crawling context's `response`. For the common case,
|
|
523
|
+
* prefer the pre-wrapped {@link handleCloudflareChallengeHook} hook.
|
|
523
524
|
*
|
|
524
525
|
* **Example usage**
|
|
525
526
|
* ```ts
|
|
526
527
|
* postNavigationHooks: [
|
|
527
|
-
* async ({ handleCloudflareChallenge })
|
|
528
|
-
* await handleCloudflareChallenge();
|
|
529
|
-
* },
|
|
528
|
+
* async (context) => ({ response: await context.handleCloudflareChallenge() }),
|
|
530
529
|
* ],
|
|
531
530
|
* ```
|
|
532
531
|
*
|
|
533
532
|
* @param [options]
|
|
534
533
|
*/
|
|
535
|
-
handleCloudflareChallenge(options?: HandleCloudflareChallengeOptions): Promise<
|
|
534
|
+
handleCloudflareChallenge(options?: HandleCloudflareChallengeOptions): Promise<Response | undefined>;
|
|
535
|
+
/**
|
|
536
|
+
* Returns the list of {@link https://playwright.dev/docs/api/class-download | Download} objects
|
|
537
|
+
* collected during the current page navigation and request handler.
|
|
538
|
+
*
|
|
539
|
+
* Useful for accessing files that the page downloads automatically.
|
|
540
|
+
* For most use cases, prefer re-enqueueing the URL to {@link FileDownload}.
|
|
541
|
+
* Use this only when direct access to the Playwright `Download` object is required.
|
|
542
|
+
*
|
|
543
|
+
* **Example usage**
|
|
544
|
+
* ```ts
|
|
545
|
+
* requestHandler: async ({ listDownloads }) => {
|
|
546
|
+
* for (const download of await listDownloads()) {
|
|
547
|
+
* try {
|
|
548
|
+
* const stream = await download.createReadStream();
|
|
549
|
+
* // stream to storage...
|
|
550
|
+
* } catch {
|
|
551
|
+
* // download failed or was cancelled
|
|
552
|
+
* }
|
|
553
|
+
* }
|
|
554
|
+
* },
|
|
555
|
+
* ```
|
|
556
|
+
*/
|
|
557
|
+
listDownloads(): Promise<Download[]>;
|
|
536
558
|
}
|
|
537
|
-
export declare function registerUtilsToContext(context: PlaywrightCrawlingContext, crawlerOptions: PlaywrightCrawlerOptions): void;
|
|
538
559
|
export { enqueueLinksByClickingElements };
|
|
539
|
-
/** @internal */
|
|
540
|
-
export declare const playwrightUtils: {
|
|
541
|
-
injectFile: typeof injectFile;
|
|
542
|
-
injectJQuery: typeof injectJQuery;
|
|
543
|
-
gotoExtended: typeof gotoExtended;
|
|
544
|
-
blockRequests: typeof blockRequests;
|
|
545
|
-
enqueueLinksByClickingElements: typeof enqueueLinksByClickingElements;
|
|
546
|
-
parseWithCheerio: typeof parseWithCheerio;
|
|
547
|
-
infiniteScroll: typeof infiniteScroll;
|
|
548
|
-
saveSnapshot: typeof saveSnapshot;
|
|
549
|
-
compileScript: typeof compileScript;
|
|
550
|
-
closeCookieModals: typeof closeCookieModals;
|
|
551
|
-
RenderingTypePredictor: typeof RenderingTypePredictor;
|
|
552
|
-
handleCloudflareChallenge: typeof handleCloudflareChallenge;
|
|
553
|
-
};
|
|
554
|
-
//# sourceMappingURL=playwright-utils.d.ts.map
|