@crawlee/playwright 4.0.0-beta.2 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +4 -3
- package/index.js +2 -2
- package/internals/adaptive-playwright-crawler.d.ts +146 -91
- package/internals/adaptive-playwright-crawler.js +485 -268
- package/internals/enqueue-links/click-elements.d.ts +37 -55
- package/internals/enqueue-links/click-elements.js +65 -55
- package/internals/playwright-browser-pool.d.ts +71 -0
- package/internals/playwright-browser-pool.js +61 -0
- package/internals/playwright-crawler.d.ts +177 -172
- package/internals/playwright-crawler.js +103 -61
- package/internals/playwright-launcher.d.ts +30 -20
- package/internals/playwright-launcher.js +22 -17
- package/internals/utils/playwright-utils.d.ts +55 -50
- package/internals/utils/playwright-utils.js +121 -148
- package/internals/utils/rendering-type-prediction.d.ts +44 -13
- package/internals/utils/rendering-type-prediction.js +95 -29
- package/package.json +15 -15
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
- package/internals/adaptive-playwright-crawler.js.map +0 -1
- package/internals/enqueue-links/click-elements.d.ts.map +0 -1
- package/internals/enqueue-links/click-elements.js.map +0 -1
- package/internals/playwright-crawler.d.ts.map +0 -1
- package/internals/playwright-crawler.js.map +0 -1
- package/internals/playwright-launcher.d.ts.map +0 -1
- package/internals/playwright-launcher.js.map +0 -1
- package/internals/utils/playwright-utils.d.ts.map +0 -1
- package/internals/utils/playwright-utils.js.map +0 -1
- package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
- package/internals/utils/rendering-type-prediction.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { IRequestManager, RequestTransform, SkippedRequestCallback, UrlPatternInput } from '@crawlee/browser';
|
|
2
2
|
import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
|
|
3
3
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
4
4
|
import type { Page } from 'playwright';
|
|
@@ -9,9 +9,9 @@ export interface EnqueueLinksByClickingElementsOptions {
|
|
|
9
9
|
*/
|
|
10
10
|
page: Page;
|
|
11
11
|
/**
|
|
12
|
-
* A request
|
|
12
|
+
* * A request manager to which the URLs will be enqueued.
|
|
13
13
|
*/
|
|
14
|
-
|
|
14
|
+
requestManager: IRequestManager;
|
|
15
15
|
/**
|
|
16
16
|
* A CSS selector matching elements to be clicked on. Unlike in {@link enqueueLinks}, there is no default
|
|
17
17
|
* value. This is to prevent suboptimal use of this function by using it too broadly.
|
|
@@ -26,72 +26,50 @@ export interface EnqueueLinksByClickingElementsOptions {
|
|
|
26
26
|
*/
|
|
27
27
|
clickOptions?: ClickOptions;
|
|
28
28
|
/**
|
|
29
|
-
* An array of
|
|
30
|
-
* containing glob pattern strings matching the URLs to be enqueued.
|
|
29
|
+
* An array of URL patterns that URLs must match to be enqueued.
|
|
31
30
|
*
|
|
32
|
-
*
|
|
33
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
31
|
+
* Accepts glob pattern strings, `{ glob: string }` objects, `RegExp` instances, or `{ regexp: RegExp }` objects.
|
|
34
32
|
*
|
|
35
|
-
*
|
|
36
|
-
* If you need case-sensitive matching, use `
|
|
33
|
+
* Glob matching is always case-insensitive.
|
|
34
|
+
* If you need case-sensitive matching, use a `RegExp`.
|
|
37
35
|
*
|
|
38
|
-
* If `
|
|
36
|
+
* If `include` is an empty array or `undefined`, then the function
|
|
39
37
|
* enqueues all the intercepted navigation requests produced by the page
|
|
40
38
|
* after clicking on elements matching the provided CSS selector.
|
|
41
39
|
*/
|
|
42
|
-
|
|
40
|
+
include?: UrlPatternInput[];
|
|
43
41
|
/**
|
|
44
|
-
* An array of
|
|
45
|
-
* containing regular expressions matching the URLs to be enqueued.
|
|
42
|
+
* An array of URL patterns. Matching URLs will **not** be enqueued.
|
|
46
43
|
*
|
|
47
|
-
*
|
|
48
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
44
|
+
* Accepts glob pattern strings, `{ glob: string }` objects, `RegExp` instances, or `{ regexp: RegExp }` objects.
|
|
49
45
|
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
* after clicking on elements matching the provided CSS selector.
|
|
53
|
-
*/
|
|
54
|
-
regexps?: RegExpInput[];
|
|
55
|
-
/**
|
|
56
|
-
* *NOTE:* In future versions of SDK the options will be removed.
|
|
57
|
-
* Please use `globs` or `regexps` instead.
|
|
58
|
-
*
|
|
59
|
-
* An array of {@link PseudoUrl} strings or plain objects
|
|
60
|
-
* containing {@link PseudoUrl} strings matching the URLs to be enqueued.
|
|
61
|
-
*
|
|
62
|
-
* The plain objects must include at least the `purl` property, which holds the pseudo-URL pattern string.
|
|
63
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
64
|
-
*
|
|
65
|
-
* With a pseudo-URL string, the matching is always case-insensitive.
|
|
66
|
-
* If you need case-sensitive matching, use `regexps` property directly.
|
|
67
|
-
*
|
|
68
|
-
* If `pseudoUrls` is an empty array or `undefined`, then the function
|
|
69
|
-
* enqueues all the intercepted navigation requests produced by the page
|
|
70
|
-
* after clicking on elements matching the provided CSS selector.
|
|
71
|
-
*
|
|
72
|
-
* @deprecated prefer using `globs` or `regexps` instead
|
|
46
|
+
* Glob matching is always case-insensitive.
|
|
47
|
+
* If you need case-sensitive matching, use a `RegExp`.
|
|
73
48
|
*/
|
|
74
|
-
|
|
49
|
+
exclude?: readonly UrlPatternInput[];
|
|
75
50
|
/**
|
|
76
|
-
*
|
|
77
|
-
* to remove
|
|
78
|
-
* when you need to enqueue multiple `Requests` to the queue that share the same URL,
|
|
79
|
-
* or to dynamically update or create `userData`.
|
|
80
|
-
*
|
|
81
|
-
* For example: by adding `useExtendedUniqueKey: true` to the `request` object, `uniqueKey` will be computed from
|
|
82
|
-
* a combination of `url`, `method` and `payload` which enables crawling of websites that navigate using form submits
|
|
83
|
-
* (POST requests).
|
|
51
|
+
* After request options are filtered by `include`/`exclude` patterns,
|
|
52
|
+
* this function can be used to remove them or modify their contents such as `userData`, `payload` or, most importantly
|
|
53
|
+
* `uniqueKey`. This is useful when you need to enqueue multiple `Requests` to the queue that share the same URL,
|
|
54
|
+
* but differ in methods or payloads, or to dynamically update or create `userData`.
|
|
84
55
|
*
|
|
85
56
|
* **Example:**
|
|
86
57
|
* ```javascript
|
|
87
58
|
* {
|
|
88
59
|
* transformRequestFunction: (request) => {
|
|
89
60
|
* request.userData.foo = 'bar';
|
|
90
|
-
* request.useExtendedUniqueKey = true;
|
|
91
61
|
* return request;
|
|
92
62
|
* }
|
|
93
63
|
* }
|
|
94
64
|
* ```
|
|
65
|
+
*
|
|
66
|
+
* Note that `transformRequestFunction` has the highest priority and can overwrite
|
|
67
|
+
* the global `label` option.
|
|
68
|
+
*
|
|
69
|
+
* The function receives a {@link RequestOptions} object and can return either:
|
|
70
|
+
* - The modified {@link RequestOptions} object
|
|
71
|
+
* - `'unchanged'` to keep the original options as-is
|
|
72
|
+
* - A falsy value or `'skip'` to exclude the request from the queue
|
|
95
73
|
*/
|
|
96
74
|
transformRequestFunction?: RequestTransform;
|
|
97
75
|
/**
|
|
@@ -131,6 +109,12 @@ export interface EnqueueLinksByClickingElementsOptions {
|
|
|
131
109
|
* @default false
|
|
132
110
|
*/
|
|
133
111
|
skipNavigation?: boolean;
|
|
112
|
+
/**
|
|
113
|
+
* When a request is skipped for some reason, you can use this callback to act on it.
|
|
114
|
+
* This is fired for requests skipped because they don't match enqueueLinks filters
|
|
115
|
+
* or because they were removed by `transformRequestFunction`.
|
|
116
|
+
*/
|
|
117
|
+
onSkippedRequest?: SkippedRequestCallback;
|
|
134
118
|
}
|
|
135
119
|
/**
|
|
136
120
|
* The function finds elements matching a specific CSS selector in a Playwright page,
|
|
@@ -141,8 +125,7 @@ export interface EnqueueLinksByClickingElementsOptions {
|
|
|
141
125
|
* in `href` elements, but rather navigations are triggered in click handlers.
|
|
142
126
|
* If you're looking to find URLs in `href` attributes of the page, see {@link enqueueLinks}.
|
|
143
127
|
*
|
|
144
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of
|
|
145
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
128
|
+
* Optionally, the function allows you to filter the target links' URLs using an array of glob or regexp patterns.
|
|
146
129
|
*
|
|
147
130
|
* **IMPORTANT**: To be able to do this, this function uses various mutations on the page,
|
|
148
131
|
* such as changing the Z-index of elements being clicked and their visibility. Therefore,
|
|
@@ -162,11 +145,11 @@ export interface EnqueueLinksByClickingElementsOptions {
|
|
|
162
145
|
* ```javascript
|
|
163
146
|
* await playwrightUtils.enqueueLinksByClickingElements({
|
|
164
147
|
* page,
|
|
165
|
-
*
|
|
148
|
+
* requestManager,
|
|
166
149
|
* selector: 'a.product-detail',
|
|
167
|
-
*
|
|
168
|
-
* 'https://www.example.com/handbags
|
|
169
|
-
* 'https://www.example.com/purses
|
|
150
|
+
* include: [
|
|
151
|
+
* 'https://www.example.com/handbags/*',
|
|
152
|
+
* 'https://www.example.com/purses/*',
|
|
170
153
|
* ],
|
|
171
154
|
* });
|
|
172
155
|
* ```
|
|
@@ -200,4 +183,3 @@ export declare function clickElementsAndInterceptNavigationRequests(options: Cli
|
|
|
200
183
|
*/
|
|
201
184
|
export declare function clickElements(page: Page, selector: string, clickOptions?: ClickOptions): Promise<void>;
|
|
202
185
|
export {};
|
|
203
|
-
//# sourceMappingURL=click-elements.d.ts.map
|
|
@@ -1,9 +1,25 @@
|
|
|
1
1
|
import { URL } from 'node:url';
|
|
2
|
-
import {
|
|
3
|
-
import
|
|
4
|
-
import
|
|
2
|
+
import { applyRequestTransform, constructUrlPatternObjects, createRequestOptions, createSkippedRequestArgs, filterRequestOptionsByPatterns, urlPatternSchema, Request as CrawleeRequest, serviceLocator, } from '@crawlee/browser';
|
|
3
|
+
import { parseArgument, schemas } from '@crawlee/utils/internal';
|
|
4
|
+
import { z } from 'zod';
|
|
5
5
|
const STARTING_Z_INDEX = 2147400000;
|
|
6
|
-
const
|
|
6
|
+
const getLog = () => serviceLocator.getChildLog('Playwright Click Elements');
|
|
7
|
+
const enqueueLinksByClickingElementsOptionsSchema = z.strictObject({
|
|
8
|
+
page: schemas.objectWithKeys(['goto', 'evaluate']),
|
|
9
|
+
requestManager: schemas.objectWithKeys(['fetchNextRequest', 'addRequestsBatched']),
|
|
10
|
+
selector: z.string(),
|
|
11
|
+
userData: schemas.anyObject.optional(),
|
|
12
|
+
clickOptions: schemas.anyObject.optional(),
|
|
13
|
+
include: schemas.arrayOf(urlPatternSchema, 'URL patterns').optional(),
|
|
14
|
+
exclude: schemas.arrayOf(urlPatternSchema, 'URL patterns').optional(),
|
|
15
|
+
transformRequestFunction: schemas.anyFunction.optional(),
|
|
16
|
+
waitForPageIdleSecs: schemas.anyNumber.default(1),
|
|
17
|
+
maxWaitForPageIdleSecs: schemas.anyNumber.default(5),
|
|
18
|
+
label: z.string().optional(),
|
|
19
|
+
forefront: z.boolean().optional(),
|
|
20
|
+
skipNavigation: z.boolean().optional(),
|
|
21
|
+
onSkippedRequest: schemas.anyFunction.optional(),
|
|
22
|
+
});
|
|
7
23
|
/**
|
|
8
24
|
* The function finds elements matching a specific CSS selector in a Playwright page,
|
|
9
25
|
* clicks all those elements using a mouse move and a left mouse button click and intercepts
|
|
@@ -13,8 +29,7 @@ const log = log_.child({ prefix: 'Playwright Click Elements' });
|
|
|
13
29
|
* in `href` elements, but rather navigations are triggered in click handlers.
|
|
14
30
|
* If you're looking to find URLs in `href` attributes of the page, see {@link enqueueLinks}.
|
|
15
31
|
*
|
|
16
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of
|
|
17
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
32
|
+
* Optionally, the function allows you to filter the target links' URLs using an array of glob or regexp patterns.
|
|
18
33
|
*
|
|
19
34
|
* **IMPORTANT**: To be able to do this, this function uses various mutations on the page,
|
|
20
35
|
* such as changing the Z-index of elements being clicked and their visibility. Therefore,
|
|
@@ -34,11 +49,11 @@ const log = log_.child({ prefix: 'Playwright Click Elements' });
|
|
|
34
49
|
* ```javascript
|
|
35
50
|
* await playwrightUtils.enqueueLinksByClickingElements({
|
|
36
51
|
* page,
|
|
37
|
-
*
|
|
52
|
+
* requestManager,
|
|
38
53
|
* selector: 'a.product-detail',
|
|
39
|
-
*
|
|
40
|
-
* 'https://www.example.com/handbags
|
|
41
|
-
* 'https://www.example.com/purses
|
|
54
|
+
* include: [
|
|
55
|
+
* 'https://www.example.com/handbags/*',
|
|
56
|
+
* 'https://www.example.com/purses/*',
|
|
42
57
|
* ],
|
|
43
58
|
* });
|
|
44
59
|
* ```
|
|
@@ -46,36 +61,13 @@ const log = log_.child({ prefix: 'Playwright Click Elements' });
|
|
|
46
61
|
* @returns Promise that resolves to {@link BatchAddRequestsResult} object.
|
|
47
62
|
*/
|
|
48
63
|
export async function enqueueLinksByClickingElements(options) {
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
requestQueue: ow.object.hasKeys('fetchNextRequest', 'addRequest'),
|
|
52
|
-
selector: ow.string,
|
|
53
|
-
userData: ow.optional.object,
|
|
54
|
-
clickOptions: ow.optional.object.hasKeys('clickCount', 'delay'),
|
|
55
|
-
pseudoUrls: ow.optional.array.ofType(ow.any(ow.string, ow.object.hasKeys('purl'))),
|
|
56
|
-
globs: ow.optional.array.ofType(ow.any(ow.string, ow.object.hasKeys('glob'))),
|
|
57
|
-
regexps: ow.optional.array.ofType(ow.any(ow.regExp, ow.object.hasKeys('regexp'))),
|
|
58
|
-
transformRequestFunction: ow.optional.function,
|
|
59
|
-
waitForPageIdleSecs: ow.optional.number,
|
|
60
|
-
maxWaitForPageIdleSecs: ow.optional.number,
|
|
61
|
-
label: ow.optional.string,
|
|
62
|
-
forefront: ow.optional.boolean,
|
|
63
|
-
skipNavigation: ow.optional.boolean,
|
|
64
|
-
}));
|
|
65
|
-
const { page, requestQueue, selector, clickOptions, pseudoUrls, globs, regexps, transformRequestFunction, waitForPageIdleSecs = 1, maxWaitForPageIdleSecs = 5, forefront, } = options;
|
|
64
|
+
const parsedOptions = parseArgument(options, enqueueLinksByClickingElementsOptionsSchema, 'EnqueueLinksByClickingElementsOptions');
|
|
65
|
+
const { page, requestManager, selector, clickOptions, include, exclude, transformRequestFunction, waitForPageIdleSecs, maxWaitForPageIdleSecs, forefront, onSkippedRequest, } = parsedOptions;
|
|
66
66
|
const waitForPageIdleMillis = waitForPageIdleSecs * 1000;
|
|
67
67
|
const maxWaitForPageIdleMillis = maxWaitForPageIdleSecs * 1000;
|
|
68
|
-
const
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
urlPatternObjects.push(...constructRegExpObjectsFromPseudoUrls(pseudoUrls));
|
|
72
|
-
}
|
|
73
|
-
if (globs?.length) {
|
|
74
|
-
urlPatternObjects.push(...constructGlobObjectsFromGlobs(globs));
|
|
75
|
-
}
|
|
76
|
-
if (regexps?.length) {
|
|
77
|
-
urlPatternObjects.push(...constructRegExpObjectsFromRegExps(regexps));
|
|
78
|
-
}
|
|
68
|
+
const hasOnSkippedRequest = onSkippedRequest !== undefined;
|
|
69
|
+
const urlExcludePatternObjects = exclude?.length ? constructUrlPatternObjects(exclude) : [];
|
|
70
|
+
const urlPatternObjects = include?.length ? constructUrlPatternObjects(include) : [];
|
|
79
71
|
const interceptedRequests = await clickElementsAndInterceptNavigationRequests({
|
|
80
72
|
page,
|
|
81
73
|
selector,
|
|
@@ -83,12 +75,21 @@ export async function enqueueLinksByClickingElements(options) {
|
|
|
83
75
|
maxWaitForPageIdleMillis,
|
|
84
76
|
clickOptions,
|
|
85
77
|
});
|
|
86
|
-
|
|
78
|
+
const requestOptions = createRequestOptions(interceptedRequests, parsedOptions);
|
|
79
|
+
const skippedByFilters = [];
|
|
80
|
+
let filteredOptions = filterRequestOptionsByPatterns(requestOptions, urlPatternObjects.length > 0 ? urlPatternObjects : undefined, urlExcludePatternObjects, undefined, hasOnSkippedRequest ? (opts) => skippedByFilters.push(opts) : undefined);
|
|
81
|
+
if (onSkippedRequest && skippedByFilters.length > 0) {
|
|
82
|
+
await Promise.all(skippedByFilters.map(async (opts) => onSkippedRequest(createSkippedRequestArgs(opts, 'filters'))));
|
|
83
|
+
}
|
|
87
84
|
if (transformRequestFunction) {
|
|
88
|
-
|
|
85
|
+
const skippedByTransform = [];
|
|
86
|
+
filteredOptions = applyRequestTransform(filteredOptions, transformRequestFunction, hasOnSkippedRequest ? (r) => skippedByTransform.push(r) : undefined);
|
|
87
|
+
if (onSkippedRequest && skippedByTransform.length > 0) {
|
|
88
|
+
await Promise.all(skippedByTransform.map(async (r) => onSkippedRequest(createSkippedRequestArgs(r, 'transform'))));
|
|
89
|
+
}
|
|
89
90
|
}
|
|
90
|
-
const requests =
|
|
91
|
-
const { addedRequests } = await
|
|
91
|
+
const requests = filteredOptions.map((opts) => new CrawleeRequest(opts));
|
|
92
|
+
const { addedRequests } = await requestManager.addRequestsBatched(requests, { forefront });
|
|
92
93
|
return { processedRequests: addedRequests, unprocessedRequests: [] };
|
|
93
94
|
}
|
|
94
95
|
/**
|
|
@@ -114,7 +115,7 @@ export async function clickElementsAndInterceptNavigationRequests(options) {
|
|
|
114
115
|
await restoreHistoryNavigationAndSaveCapturedUrls(page, uniqueRequests);
|
|
115
116
|
// browser.off(BrowserEmittedEvents.TargetCreated, onTargetCreated);
|
|
116
117
|
page.off('framenavigated', onFrameNavigated);
|
|
117
|
-
await context.unroute('
|
|
118
|
+
await context.unroute('**', onInterceptedRequest);
|
|
118
119
|
const serializedRequests = Array.from(uniqueRequests);
|
|
119
120
|
return serializedRequests.map((r) => JSON.parse(r));
|
|
120
121
|
}
|
|
@@ -150,7 +151,9 @@ function createTargetCreatedHandler(requests) {
|
|
|
150
151
|
await popup.close();
|
|
151
152
|
}
|
|
152
153
|
catch (err) {
|
|
153
|
-
|
|
154
|
+
getLog().debug('enqueueLinksByClickingElements: Could not close spawned page.', {
|
|
155
|
+
error: err.stack,
|
|
156
|
+
});
|
|
154
157
|
}
|
|
155
158
|
};
|
|
156
159
|
}
|
|
@@ -158,7 +161,16 @@ function createTargetCreatedHandler(requests) {
|
|
|
158
161
|
* @ignore
|
|
159
162
|
*/
|
|
160
163
|
function isTopFrameNavigationRequest(page, req) {
|
|
161
|
-
|
|
164
|
+
try {
|
|
165
|
+
return req.isNavigationRequest() && req.frame() === page.mainFrame();
|
|
166
|
+
}
|
|
167
|
+
catch {
|
|
168
|
+
// `req.frame()` throws when the owning frame is unavailable - e.g. the request was
|
|
169
|
+
// issued by a service worker, or before/after its frame existed (see #3216). Such a
|
|
170
|
+
// request is not a top-frame navigation, so swallow the throw and let it pass through
|
|
171
|
+
// instead of crashing the route handler (which would leave the route unhandled).
|
|
172
|
+
return false;
|
|
173
|
+
}
|
|
162
174
|
}
|
|
163
175
|
/**
|
|
164
176
|
* @ignore
|
|
@@ -223,7 +235,7 @@ function updateElementCssToEnableMouseClick(el, zIndex) {
|
|
|
223
235
|
*/
|
|
224
236
|
export async function clickElements(page, selector, clickOptions) {
|
|
225
237
|
const elementHandles = await page.$$(selector);
|
|
226
|
-
|
|
238
|
+
getLog().debug(`enqueueLinksByClickingElements: There are ${elementHandles.length} elements to click.`);
|
|
227
239
|
let clickedElementsCount = 0;
|
|
228
240
|
let zIndex = STARTING_Z_INDEX;
|
|
229
241
|
let shouldLogWarning = true;
|
|
@@ -236,15 +248,15 @@ export async function clickElements(page, selector, clickOptions) {
|
|
|
236
248
|
catch (err) {
|
|
237
249
|
const e = err;
|
|
238
250
|
if (shouldLogWarning && e.stack.includes('is detached from document')) {
|
|
239
|
-
|
|
251
|
+
getLog().warning(`An element with selector ${selector} that you're trying to click has been removed from the page. ` +
|
|
240
252
|
'This was probably caused by an earlier click which triggered some JavaScript on the page that caused it to change. ' +
|
|
241
253
|
'If you\'re trying to enqueue pagination links, we suggest using the "next" button, if available and going one by one.');
|
|
242
254
|
shouldLogWarning = false;
|
|
243
255
|
}
|
|
244
|
-
|
|
256
|
+
getLog().debug('enqueueLinksByClickingElements: Click failed.', { stack: e.stack });
|
|
245
257
|
}
|
|
246
258
|
}
|
|
247
|
-
|
|
259
|
+
getLog().debug(`enqueueLinksByClickingElements: Successfully clicked ${clickedElementsCount} elements out of ${elementHandles.length}`);
|
|
248
260
|
}
|
|
249
261
|
/**
|
|
250
262
|
* This function tracks whether any requests, frame navigations or targets were emitted
|
|
@@ -263,8 +275,6 @@ export async function clickElements(page, selector, clickOptions) {
|
|
|
263
275
|
async function waitForPageIdle({ page, waitForPageIdleMillis, maxWaitForPageIdleMillis, }) {
|
|
264
276
|
return new Promise((resolve) => {
|
|
265
277
|
let timeout;
|
|
266
|
-
let maxTimeout;
|
|
267
|
-
page.on('popup', activityHandler);
|
|
268
278
|
function activityHandler() {
|
|
269
279
|
clearTimeout(timeout);
|
|
270
280
|
timeout = setTimeout(() => {
|
|
@@ -273,7 +283,7 @@ async function waitForPageIdle({ page, waitForPageIdleMillis, maxWaitForPageIdle
|
|
|
273
283
|
}, waitForPageIdleMillis);
|
|
274
284
|
}
|
|
275
285
|
function maxTimeoutHandler() {
|
|
276
|
-
|
|
286
|
+
getLog().debug(`enqueueLinksByClickingElements: Page still showed activity after ${maxWaitForPageIdleMillis}ms. ` +
|
|
277
287
|
'This is probably due to the website itself dispatching requests, but some links may also have been missed.');
|
|
278
288
|
finish();
|
|
279
289
|
}
|
|
@@ -281,7 +291,8 @@ async function waitForPageIdle({ page, waitForPageIdleMillis, maxWaitForPageIdle
|
|
|
281
291
|
page.off('request', activityHandler).off('framenavigated', activityHandler).off('popup', activityHandler);
|
|
282
292
|
resolve();
|
|
283
293
|
}
|
|
284
|
-
maxTimeout = setTimeout(maxTimeoutHandler, maxWaitForPageIdleMillis);
|
|
294
|
+
const maxTimeout = setTimeout(maxTimeoutHandler, maxWaitForPageIdleMillis);
|
|
295
|
+
page.on('popup', activityHandler);
|
|
285
296
|
activityHandler(); // We call this once manually in case there would be no requests at all.
|
|
286
297
|
page.on('request', activityHandler);
|
|
287
298
|
page.on('framenavigated', activityHandler);
|
|
@@ -304,8 +315,7 @@ async function restoreHistoryNavigationAndSaveCapturedUrls(page, requests) {
|
|
|
304
315
|
requests.add(JSON.stringify({ url }));
|
|
305
316
|
}
|
|
306
317
|
catch (err) {
|
|
307
|
-
|
|
318
|
+
getLog().debug('enqueueLinksByClickingElements: Failed to ', { error: err.stack });
|
|
308
319
|
}
|
|
309
320
|
});
|
|
310
321
|
}
|
|
311
|
-
//# sourceMappingURL=click-elements.js.map
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import type { Configuration } from '@crawlee/browser';
|
|
2
|
+
import type { BrowserPool, BrowserPoolHooks, BrowserPoolOptions, PlaywrightPlugin, RemoteBrowserPool, RemoteBrowserPoolOptions } from '@crawlee/browser-pool';
|
|
3
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
4
|
+
import type { Page } from 'playwright';
|
|
5
|
+
import type { PlaywrightLaunchContext } from './playwright-launcher.js';
|
|
6
|
+
/** A {@link BrowserPool} of Playwright browsers, as built by {@link playwrightBrowserPool}. */
|
|
7
|
+
export type PlaywrightBrowserPool = BrowserPool<{
|
|
8
|
+
browserPlugins: [PlaywrightPlugin];
|
|
9
|
+
}, [PlaywrightPlugin]>;
|
|
10
|
+
export interface PlaywrightBrowserPoolOptions extends Omit<BrowserPoolOptions, 'browserPlugins'>, BrowserPoolHooks<ReturnType<PlaywrightPlugin['createController']>, ReturnType<PlaywrightPlugin['createLaunchContext']>, Page> {
|
|
11
|
+
/** How to launch the browser: which Playwright browser type, proxy, user data dir, ... */
|
|
12
|
+
launchContext?: PlaywrightLaunchContext;
|
|
13
|
+
/**
|
|
14
|
+
* Whether to run the browser in headless mode. Shorthand for `launchContext.launchOptions.headless`.
|
|
15
|
+
* Defaults to `true`, and can also be set via {@link Configuration}.
|
|
16
|
+
*/
|
|
17
|
+
headless?: boolean;
|
|
18
|
+
/** Configuration to read the browser defaults from. Defaults to the global configuration. */
|
|
19
|
+
configuration?: Configuration;
|
|
20
|
+
}
|
|
21
|
+
export interface RemotePlaywrightBrowserPoolOptions extends Pick<PlaywrightBrowserPoolOptions, 'launchContext' | 'headless' | 'configuration'>, Omit<RemoteBrowserPoolOptions, 'browserPlugins'> {
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Builds a {@link BrowserPool} of Playwright browsers to pass to a {@link PlaywrightCrawler} as its
|
|
25
|
+
* {@link BrowserCrawlerOptions.browserPool|`browserPool`}.
|
|
26
|
+
*
|
|
27
|
+
* It accepts every {@link BrowserPoolOptions|`BrowserPool` option} plus the crawler's own `launchContext` and
|
|
28
|
+
* `headless`, and derives the browser plugin from them - so a pool built here always matches the crawler it is
|
|
29
|
+
* given to, and configuring one never means assembling a {@link PlaywrightPlugin} by hand.
|
|
30
|
+
*
|
|
31
|
+
* **Example usage:**
|
|
32
|
+
*
|
|
33
|
+
* ```javascript
|
|
34
|
+
* const crawler = new PlaywrightCrawler({
|
|
35
|
+
* browserPool: playwrightBrowserPool({
|
|
36
|
+
* maxOpenPagesPerBrowser: 1,
|
|
37
|
+
* launchContext: { launcher: firefox },
|
|
38
|
+
* }),
|
|
39
|
+
* requestHandler: async ({ page }) => { ... },
|
|
40
|
+
* });
|
|
41
|
+
* ```
|
|
42
|
+
*
|
|
43
|
+
* The returned pool is *not* torn down by the crawler, which is what makes it shareable between crawlers.
|
|
44
|
+
*
|
|
45
|
+
* @category Browser management
|
|
46
|
+
*/
|
|
47
|
+
export declare function playwrightBrowserPool(options?: PlaywrightBrowserPoolOptions): PlaywrightBrowserPool;
|
|
48
|
+
/**
|
|
49
|
+
* The {@link RemoteBrowserPool} counterpart of {@link playwrightBrowserPool}: connects to a remote browser
|
|
50
|
+
* service (Browserbase, Browserless, Steel, ...) with a Playwright plugin derived from `launchContext`.
|
|
51
|
+
*
|
|
52
|
+
* A {@link PlaywrightCrawler} accepts the same connection details directly via
|
|
53
|
+
* {@link BrowserCrawlerOptions.remoteBrowser|`remoteBrowser`}; reach for this factory when you also need to
|
|
54
|
+
* tune the wrapping pool, or to share one remote pool between crawlers.
|
|
55
|
+
*
|
|
56
|
+
* **Example usage:**
|
|
57
|
+
*
|
|
58
|
+
* ```javascript
|
|
59
|
+
* const crawler = new PlaywrightCrawler({
|
|
60
|
+
* browserPool: remotePlaywrightBrowserPool({
|
|
61
|
+
* endpoint: 'wss://production-sfo.browserless.io?token=xxx',
|
|
62
|
+
* maxOpenBrowsers: 2,
|
|
63
|
+
* browserPoolOptions: { useFingerprints: false },
|
|
64
|
+
* }),
|
|
65
|
+
* requestHandler: async ({ page }) => { ... },
|
|
66
|
+
* });
|
|
67
|
+
* ```
|
|
68
|
+
*
|
|
69
|
+
* @category Browser management
|
|
70
|
+
*/
|
|
71
|
+
export declare function remotePlaywrightBrowserPool(options: RemotePlaywrightBrowserPoolOptions): RemoteBrowserPool<Page>;
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { PlaywrightLauncher } from './playwright-launcher.js';
|
|
2
|
+
/**
|
|
3
|
+
* Builds a {@link BrowserPool} of Playwright browsers to pass to a {@link PlaywrightCrawler} as its
|
|
4
|
+
* {@link BrowserCrawlerOptions.browserPool|`browserPool`}.
|
|
5
|
+
*
|
|
6
|
+
* It accepts every {@link BrowserPoolOptions|`BrowserPool` option} plus the crawler's own `launchContext` and
|
|
7
|
+
* `headless`, and derives the browser plugin from them - so a pool built here always matches the crawler it is
|
|
8
|
+
* given to, and configuring one never means assembling a {@link PlaywrightPlugin} by hand.
|
|
9
|
+
*
|
|
10
|
+
* **Example usage:**
|
|
11
|
+
*
|
|
12
|
+
* ```javascript
|
|
13
|
+
* const crawler = new PlaywrightCrawler({
|
|
14
|
+
* browserPool: playwrightBrowserPool({
|
|
15
|
+
* maxOpenPagesPerBrowser: 1,
|
|
16
|
+
* launchContext: { launcher: firefox },
|
|
17
|
+
* }),
|
|
18
|
+
* requestHandler: async ({ page }) => { ... },
|
|
19
|
+
* });
|
|
20
|
+
* ```
|
|
21
|
+
*
|
|
22
|
+
* The returned pool is *not* torn down by the crawler, which is what makes it shareable between crawlers.
|
|
23
|
+
*
|
|
24
|
+
* @category Browser management
|
|
25
|
+
*/
|
|
26
|
+
export function playwrightBrowserPool(options = {}) {
|
|
27
|
+
const { launchContext, headless, configuration, ...poolOptions } = options;
|
|
28
|
+
return playwrightLauncher(launchContext, headless, configuration).createBrowserPool(poolOptions);
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* The {@link RemoteBrowserPool} counterpart of {@link playwrightBrowserPool}: connects to a remote browser
|
|
32
|
+
* service (Browserbase, Browserless, Steel, ...) with a Playwright plugin derived from `launchContext`.
|
|
33
|
+
*
|
|
34
|
+
* A {@link PlaywrightCrawler} accepts the same connection details directly via
|
|
35
|
+
* {@link BrowserCrawlerOptions.remoteBrowser|`remoteBrowser`}; reach for this factory when you also need to
|
|
36
|
+
* tune the wrapping pool, or to share one remote pool between crawlers.
|
|
37
|
+
*
|
|
38
|
+
* **Example usage:**
|
|
39
|
+
*
|
|
40
|
+
* ```javascript
|
|
41
|
+
* const crawler = new PlaywrightCrawler({
|
|
42
|
+
* browserPool: remotePlaywrightBrowserPool({
|
|
43
|
+
* endpoint: 'wss://production-sfo.browserless.io?token=xxx',
|
|
44
|
+
* maxOpenBrowsers: 2,
|
|
45
|
+
* browserPoolOptions: { useFingerprints: false },
|
|
46
|
+
* }),
|
|
47
|
+
* requestHandler: async ({ page }) => { ... },
|
|
48
|
+
* });
|
|
49
|
+
* ```
|
|
50
|
+
*
|
|
51
|
+
* @category Browser management
|
|
52
|
+
*/
|
|
53
|
+
export function remotePlaywrightBrowserPool(options) {
|
|
54
|
+
const { launchContext, headless, configuration, ...remoteOptions } = options;
|
|
55
|
+
return playwrightLauncher(launchContext, headless, configuration).createRemoteBrowserPool(remoteOptions);
|
|
56
|
+
}
|
|
57
|
+
function playwrightLauncher(launchContext = {}, headless, configuration) {
|
|
58
|
+
return new PlaywrightLauncher(headless == null
|
|
59
|
+
? launchContext
|
|
60
|
+
: { ...launchContext, launchOptions: { ...launchContext.launchOptions, headless } }, configuration);
|
|
61
|
+
}
|