@crawlee/playwright 4.0.0-beta.2 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +4 -3
- package/index.js +2 -2
- package/internals/adaptive-playwright-crawler.d.ts +146 -91
- package/internals/adaptive-playwright-crawler.js +485 -268
- package/internals/enqueue-links/click-elements.d.ts +37 -55
- package/internals/enqueue-links/click-elements.js +65 -55
- package/internals/playwright-browser-pool.d.ts +71 -0
- package/internals/playwright-browser-pool.js +61 -0
- package/internals/playwright-crawler.d.ts +177 -172
- package/internals/playwright-crawler.js +103 -61
- package/internals/playwright-launcher.d.ts +30 -20
- package/internals/playwright-launcher.js +22 -17
- package/internals/utils/playwright-utils.d.ts +55 -50
- package/internals/utils/playwright-utils.js +121 -148
- package/internals/utils/rendering-type-prediction.d.ts +44 -13
- package/internals/utils/rendering-type-prediction.js +95 -29
- package/package.json +15 -15
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
- package/internals/adaptive-playwright-crawler.js.map +0 -1
- package/internals/enqueue-links/click-elements.d.ts.map +0 -1
- package/internals/enqueue-links/click-elements.js.map +0 -1
- package/internals/playwright-crawler.d.ts.map +0 -1
- package/internals/playwright-crawler.js.map +0 -1
- package/internals/playwright-launcher.d.ts.map +0 -1
- package/internals/playwright-launcher.js.map +0 -1
- package/internals/utils/playwright-utils.d.ts.map +0 -1
- package/internals/utils/playwright-utils.js.map +0 -1
- package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
- package/internals/utils/rendering-type-prediction.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
package/README.md
CHANGED
|
@@ -1,19 +1,23 @@
|
|
|
1
1
|
<h1 align="center">
|
|
2
2
|
<a href="https://crawlee.dev">
|
|
3
3
|
<picture>
|
|
4
|
-
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true"
|
|
5
|
-
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500"
|
|
4
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true" />
|
|
5
|
+
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500" />
|
|
6
6
|
</picture>
|
|
7
7
|
</a>
|
|
8
|
-
<br
|
|
8
|
+
<br />
|
|
9
9
|
<small>A web scraping and browser automation library</small>
|
|
10
10
|
</h1>
|
|
11
11
|
|
|
12
|
-
<p align=center>
|
|
13
|
-
<a href="https://
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
12
|
+
<p align="center">
|
|
13
|
+
<a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" width="250" height="55"/></a>
|
|
14
|
+
</p>
|
|
15
|
+
|
|
16
|
+
<p align="center">
|
|
17
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" /></a>
|
|
18
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" /></a>
|
|
19
|
+
<a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" /></a>
|
|
20
|
+
<a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" /></a>
|
|
17
21
|
</p>
|
|
18
22
|
|
|
19
23
|
Crawlee covers your crawling and scraping end-to-end and **helps you build reliable scrapers. Fast.**
|
|
@@ -24,7 +28,7 @@ Crawlee is available as the [`crawlee`](https://www.npmjs.com/package/crawlee) N
|
|
|
24
28
|
|
|
25
29
|
> 👉 **View full documentation, guides and examples on the [Crawlee project website](https://crawlee.dev)** 👈
|
|
26
30
|
|
|
27
|
-
>
|
|
31
|
+
> Do you prefer 🐍 Python instead of JavaScript? [👉 Checkout Crawlee for Python 👈](https://github.com/apify/crawlee-python).
|
|
28
32
|
|
|
29
33
|
## Installation
|
|
30
34
|
|
|
@@ -85,7 +89,7 @@ By default, Crawlee stores data to `./storage` in the current working directory.
|
|
|
85
89
|
We provide automated beta builds for every merged code change in Crawlee. You can find them in the npm [list of releases](https://www.npmjs.com/package/crawlee?activeTab=versions). If you want to test new features or bug fixes before we release them, feel free to install a beta build like this:
|
|
86
90
|
|
|
87
91
|
```bash
|
|
88
|
-
npm install crawlee@
|
|
92
|
+
npm install crawlee@next
|
|
89
93
|
```
|
|
90
94
|
|
|
91
95
|
If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need to specify dependency overrides in your `package.json` file so that you don't end up with multiple versions of Crawlee installed:
|
|
@@ -94,9 +98,9 @@ If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need
|
|
|
94
98
|
{
|
|
95
99
|
"overrides": {
|
|
96
100
|
"apify": {
|
|
97
|
-
"@crawlee/core": "
|
|
98
|
-
"@crawlee/types": "
|
|
99
|
-
"@crawlee/utils": "
|
|
101
|
+
"@crawlee/core": "$crawlee",
|
|
102
|
+
"@crawlee/types": "$crawlee",
|
|
103
|
+
"@crawlee/utils": "$crawlee"
|
|
100
104
|
}
|
|
101
105
|
}
|
|
102
106
|
}
|
package/index.d.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
export * from '@crawlee/browser';
|
|
2
|
+
export * from './internals/playwright-browser-pool.js';
|
|
2
3
|
export * from './internals/playwright-crawler.js';
|
|
3
|
-
export
|
|
4
|
+
export { launchPlaywright } from './internals/playwright-launcher.js';
|
|
5
|
+
export type { PlaywrightLaunchContext } from './internals/playwright-launcher.js';
|
|
4
6
|
export * from './internals/adaptive-playwright-crawler.js';
|
|
5
7
|
export { RenderingTypePredictor } from './internals/utils/rendering-type-prediction.js';
|
|
6
8
|
export * as playwrightUtils from './internals/utils/playwright-utils.js';
|
|
7
9
|
export * as playwrightClickElements from './internals/enqueue-links/click-elements.js';
|
|
8
10
|
export type { DirectNavigationOptions as PlaywrightDirectNavigationOptions } from './internals/utils/playwright-utils.js';
|
|
9
|
-
export type { RenderingType } from './internals/utils/rendering-type-prediction.js';
|
|
10
|
-
//# sourceMappingURL=index.d.ts.map
|
|
11
|
+
export type { IRenderingTypePredictor, RenderingType } from './internals/utils/rendering-type-prediction.js';
|
package/index.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
export * from '@crawlee/browser';
|
|
2
|
+
export * from './internals/playwright-browser-pool.js';
|
|
2
3
|
export * from './internals/playwright-crawler.js';
|
|
3
|
-
export
|
|
4
|
+
export { launchPlaywright } from './internals/playwright-launcher.js';
|
|
4
5
|
export * from './internals/adaptive-playwright-crawler.js';
|
|
5
6
|
export { RenderingTypePredictor } from './internals/utils/rendering-type-prediction.js';
|
|
6
7
|
export * as playwrightUtils from './internals/utils/playwright-utils.js';
|
|
7
8
|
export * as playwrightClickElements from './internals/enqueue-links/click-elements.js';
|
|
8
|
-
//# sourceMappingURL=index.js.map
|
|
@@ -1,52 +1,58 @@
|
|
|
1
|
-
import type { BrowserHook,
|
|
2
|
-
import type {
|
|
3
|
-
import {
|
|
4
|
-
import type {
|
|
5
|
-
import {
|
|
6
|
-
import { type Cheerio } from 'cheerio';
|
|
1
|
+
import type { BrowserHook, LoadedRequest, CrawlingRequest, RouterHandler, RouteSchemas, RoutesFromSchemas } from '@crawlee/browser';
|
|
2
|
+
import type { BasicCrawlerOptions, CrawlingContext, EnqueueLinksOptions, GetUserDataFromRequest, RouterRoutes } from '@crawlee/basic';
|
|
3
|
+
import { BasicCrawler } from '@crawlee/basic';
|
|
4
|
+
import type { StorageTransactionView } from '@crawlee/core';
|
|
5
|
+
import type { Dictionary, Awaitable } from '@crawlee/types';
|
|
6
|
+
import { type Cheerio, type CheerioAPI } from 'cheerio';
|
|
7
|
+
import type { AnyNode } from 'domhandler';
|
|
7
8
|
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
8
9
|
import type { Page } from 'playwright';
|
|
9
|
-
import
|
|
10
|
+
import { z } from 'zod';
|
|
10
11
|
import type { PlaywrightCrawlerOptions, PlaywrightCrawlingContext, PlaywrightGotoOptions } from './playwright-crawler.js';
|
|
11
|
-
import {
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
12
|
+
import { type IRenderingTypePredictor } from './utils/rendering-type-prediction.js';
|
|
13
|
+
declare const adaptiveStatisticStateSchema: z.ZodObject<{
|
|
14
|
+
httpOnlyRequestHandlerRuns: z.ZodDefault<z.ZodNumber>;
|
|
15
|
+
browserRequestHandlerRuns: z.ZodDefault<z.ZodNumber>;
|
|
16
|
+
renderingTypeMispredictions: z.ZodDefault<z.ZodNumber>;
|
|
17
|
+
}, z.core.$strip>;
|
|
18
|
+
/**
|
|
19
|
+
* The extra statistics fields {@link AdaptivePlaywrightCrawler} tracks on top of the built-in
|
|
20
|
+
* {@link StatisticState} ones. They are available on `crawler.statistics.state` and are persisted with the rest of
|
|
21
|
+
* the statistics.
|
|
22
|
+
*/
|
|
23
|
+
export type AdaptivePlaywrightCrawlerStatisticState = z.infer<typeof adaptiveStatisticStateSchema>;
|
|
24
|
+
/**
|
|
25
|
+
* The {@link AdaptivePlaywrightCrawlerStatisticState} fields as a {@link Statistics} state extension, defaults
|
|
26
|
+
* and all. A {@link Statistics} instance to be injected into an {@link AdaptivePlaywrightCrawler} has to carry
|
|
27
|
+
* them - `deserialize.extend()` your own fields onto this one and pass the result as `stateExtension`.
|
|
28
|
+
*/
|
|
29
|
+
export declare const adaptivePlaywrightCrawlerStatisticState: {
|
|
30
|
+
deserialize: z.ZodObject<{
|
|
31
|
+
httpOnlyRequestHandlerRuns: z.ZodDefault<z.ZodNumber>;
|
|
32
|
+
browserRequestHandlerRuns: z.ZodDefault<z.ZodNumber>;
|
|
33
|
+
renderingTypeMispredictions: z.ZodDefault<z.ZodNumber>;
|
|
34
|
+
}, z.core.$strip>;
|
|
21
35
|
};
|
|
22
|
-
interface
|
|
23
|
-
|
|
24
|
-
browserRequestHandlerRuns?: number;
|
|
25
|
-
renderingTypeMispredictions?: number;
|
|
26
|
-
}
|
|
27
|
-
declare class AdaptivePlaywrightCrawlerStatistics extends Statistics {
|
|
28
|
-
state: AdaptivePlaywrightCrawlerStatisticState;
|
|
29
|
-
constructor(options?: StatisticsOptions);
|
|
30
|
-
reset(): void;
|
|
31
|
-
protected _maybeLoadStatistics(): Promise<void>;
|
|
32
|
-
trackHttpOnlyRequestHandlerRun(): void;
|
|
33
|
-
trackBrowserRequestHandlerRun(): void;
|
|
34
|
-
trackRenderingTypeMisprediction(): void;
|
|
35
|
-
}
|
|
36
|
-
export interface AdaptivePlaywrightCrawlerContext<UserData extends Dictionary = Dictionary> extends RestrictedCrawlingContext<UserData> {
|
|
36
|
+
export interface AdaptivePlaywrightCrawlerContext<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
|
|
37
|
+
request: LoadedRequest<CrawlingRequest<UserData>>;
|
|
37
38
|
/**
|
|
38
39
|
* The HTTP response, either from the HTTP client or from the initial request from playwright's navigation.
|
|
39
40
|
*/
|
|
40
|
-
response:
|
|
41
|
+
response: Response;
|
|
41
42
|
/**
|
|
42
43
|
* Playwright Page object. If accessed in HTTP-only rendering, this will throw an error and make the AdaptivePlaywrightCrawlerContext retry the request in a browser.
|
|
43
44
|
*/
|
|
44
45
|
page: Page;
|
|
45
46
|
/**
|
|
46
|
-
* Wait for an element matching the selector to appear and return a Cheerio object of matched
|
|
47
|
+
* Wait for an element matching the selector to appear and return a Cheerio object of the first matched element.
|
|
47
48
|
* Timeout defaults to 5s.
|
|
48
49
|
*/
|
|
49
|
-
querySelector
|
|
50
|
+
querySelector(selector: string, timeoutMs?: number): Promise<Cheerio<AnyNode>>;
|
|
51
|
+
/**
|
|
52
|
+
* Wait for an element matching the selector to appear and return a Cheerio object of all matched elements.
|
|
53
|
+
* Timeout defaults to 5s.
|
|
54
|
+
*/
|
|
55
|
+
querySelectorAll(selector: string, timeoutMs?: number): Promise<Cheerio<AnyNode>>;
|
|
50
56
|
/**
|
|
51
57
|
* Wait for an element matching the selector to appear.
|
|
52
58
|
* Timeout defaults to 5s.
|
|
@@ -73,67 +79,79 @@ export interface AdaptivePlaywrightCrawlerContext<UserData extends Dictionary =
|
|
|
73
79
|
* });
|
|
74
80
|
* ```
|
|
75
81
|
*/
|
|
76
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<
|
|
82
|
+
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
83
|
+
enqueueLinks(options?: EnqueueLinksOptions): Promise<unknown>;
|
|
77
84
|
}
|
|
78
|
-
interface
|
|
85
|
+
interface AdaptiveHookContext extends Pick<AdaptivePlaywrightCrawlerContext, 'id' | 'session' | 'proxyInfo' | 'log'> {
|
|
79
86
|
page?: Page;
|
|
80
|
-
|
|
87
|
+
request: CrawlingRequest;
|
|
88
|
+
gotoOptions?: PlaywrightGotoOptions;
|
|
81
89
|
}
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
* other than the methods of the crawling context. Any other side effects may be invoked repeatedly by the crawler, which can lead to inconsistent results.
|
|
88
|
-
*
|
|
89
|
-
* The function must return a promise, which is then awaited by the crawler.
|
|
90
|
-
*
|
|
91
|
-
* If the function throws an exception, the crawler will try to re-crawl the
|
|
92
|
-
* request later, up to `option.maxRequestRetries` times.
|
|
93
|
-
*/
|
|
94
|
-
requestHandler?: (crawlingContext: LoadedContext<AdaptivePlaywrightCrawlerContext>) => Awaitable<void>;
|
|
90
|
+
type AdaptiveHook<ContextExtension = Dictionary<never>> = BrowserHook<AdaptiveHookContext, ContextExtension>;
|
|
91
|
+
type AdaptivePostNavigationHook<ContextExtension = Dictionary<never>> = BrowserHook<Omit<AdaptiveHookContext, 'request'> & {
|
|
92
|
+
request: LoadedRequest<CrawlingRequest>;
|
|
93
|
+
}, ContextExtension>;
|
|
94
|
+
export interface AdaptivePlaywrightCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<AdaptivePlaywrightCrawlerContext['request']>>, StatisticStateExtension extends AdaptivePlaywrightCrawlerStatisticState = AdaptivePlaywrightCrawlerStatisticState> extends Omit<BasicCrawlerOptions<AdaptivePlaywrightCrawlerContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'preNavigationHooks' | 'postNavigationHooks' | 'contextPipelineBuilder'>, Pick<PlaywrightCrawlerOptions, 'launchContext' | 'headless' | 'browserPool' | 'remoteBrowser'> {
|
|
95
95
|
/**
|
|
96
96
|
* Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies.
|
|
97
97
|
* The function accepts a subset of the crawling context. If you attempt to access the `page` property during HTTP-only crawling,
|
|
98
98
|
* an exception will be thrown. If it's not caught, the request will be transparently retried in a browser.
|
|
99
|
+
*
|
|
100
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context,
|
|
101
|
+
* allowing the hook to override context members for subsequent hooks and pipeline stages.
|
|
99
102
|
*/
|
|
100
|
-
preNavigationHooks?: AdaptiveHook[];
|
|
103
|
+
preNavigationHooks?: AdaptiveHook<ContextExtension>[];
|
|
101
104
|
/**
|
|
102
105
|
* Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
|
|
103
106
|
* The function accepts a subset of the crawling context. If you attempt to access the `page` property during HTTP-only crawling,
|
|
104
107
|
* an exception will be thrown. If it's not caught, the request will be transparently retried in a browser.
|
|
108
|
+
*
|
|
109
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context
|
|
110
|
+
* (e.g. to override `response` after solving a challenge).
|
|
105
111
|
*/
|
|
106
|
-
postNavigationHooks?:
|
|
112
|
+
postNavigationHooks?: AdaptivePostNavigationHook<ContextExtension>[];
|
|
107
113
|
/**
|
|
108
114
|
* Specifies the frequency of rendering type detection checks - 0.1 means roughly 10% of requests.
|
|
109
115
|
* Defaults to 0.1 (so 10%).
|
|
110
116
|
*/
|
|
111
117
|
renderingTypeDetectionRatio?: number;
|
|
112
118
|
/**
|
|
113
|
-
* An optional callback that is called on
|
|
119
|
+
* An optional callback that is called on the storage writes recorded by the request handler in plain
|
|
120
|
+
* HTTP mode (exposed as a read-only {@link StorageTransactionView}).
|
|
114
121
|
* If it returns false, the request is retried in a browser.
|
|
115
|
-
* If no callback is specified, every
|
|
122
|
+
* If no callback is specified, every result is considered valid.
|
|
116
123
|
*/
|
|
117
|
-
resultChecker?: (result:
|
|
124
|
+
resultChecker?: (result: StorageTransactionView) => boolean;
|
|
125
|
+
/**
|
|
126
|
+
* An optional callback that decides whether an error thrown during the plain HTTP request handler
|
|
127
|
+
* should be propagated (instead of falling back to browser navigation).
|
|
128
|
+
*
|
|
129
|
+
* If the callback returns `true`, the error is thrown, triggering the standard retry mechanism.
|
|
130
|
+
* If the callback returns `false` (or is not provided), the error is logged and the crawler
|
|
131
|
+
* falls back to browser navigation (default behavior).
|
|
132
|
+
*
|
|
133
|
+
* @default () => false
|
|
134
|
+
*/
|
|
135
|
+
shouldPropagateError?: (error: Error, context: PlaywrightCrawlingContext) => Awaitable<boolean>;
|
|
118
136
|
/**
|
|
119
137
|
* An optional callback used in rendering type detection. On each detection, the result of the plain HTTP run is compared to that of the browser one.
|
|
120
|
-
* If
|
|
138
|
+
* If a callback is provided, the contract is as follows:
|
|
139
|
+
* It the callback returns true or 'equal', the results are considered equal and the target site is considered static.
|
|
140
|
+
* If it returns false or 'different', the target site is considered client-rendered.
|
|
141
|
+
* If it returns 'inconclusive', the detection result won't be used.
|
|
121
142
|
* If no result comparator is specified, but there is a `resultChecker`, any site where the `resultChecker` returns true is considered static.
|
|
122
143
|
* If neither `resultComparator` nor `resultChecker` are specified, a deep comparison of returned dataset items is used as a default.
|
|
144
|
+
*
|
|
145
|
+
* For a stricter, ready-made comparator that also takes enqueued requests and key-value store changes into account, see {@link fullResultComparator}.
|
|
123
146
|
*/
|
|
124
|
-
resultComparator?: (resultA:
|
|
125
|
-
/**
|
|
126
|
-
* A custom rendering type predictor
|
|
127
|
-
*/
|
|
128
|
-
renderingTypePredictor?: Pick<RenderingTypePredictor, 'predict' | 'storeResult'>;
|
|
147
|
+
resultComparator?: (resultA: StorageTransactionView, resultB: StorageTransactionView) => boolean | 'equal' | 'different' | 'inconclusive';
|
|
129
148
|
/**
|
|
130
|
-
*
|
|
131
|
-
*
|
|
149
|
+
* A custom rendering type predictor. A predictor passed here is borrowed - the crawler never drives its
|
|
150
|
+
* lifecycle, so set it up yourself (the built-in {@link RenderingTypePredictor} needs `initialize()`).
|
|
151
|
+
* Omit the option and the crawler builds its own from `renderingTypeDetectionRatio` - and initializes it.
|
|
132
152
|
*/
|
|
133
|
-
|
|
153
|
+
renderingTypePredictor?: IRenderingTypePredictor;
|
|
134
154
|
}
|
|
135
|
-
declare const proxyLogMethods: readonly ["error", "exception", "softFail", "info", "debug", "perf", "warningOnce", "deprecated"];
|
|
136
|
-
type LogProxyCall = [log: Log, method: (typeof proxyLogMethods)[number], ...args: unknown[]];
|
|
137
155
|
/**
|
|
138
156
|
* An extension of {@link PlaywrightCrawler} that uses a more limited request handler interface so that it is able to switch to HTTP-only crawling when it detects it may be possible.
|
|
139
157
|
*
|
|
@@ -163,31 +181,68 @@ type LogProxyCall = [log: Log, method: (typeof proxyLogMethods)[number], ...args
|
|
|
163
181
|
*
|
|
164
182
|
* @experimental
|
|
165
183
|
*/
|
|
166
|
-
export declare class AdaptivePlaywrightCrawler extends
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
private
|
|
171
|
-
private
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
*
|
|
176
|
-
*
|
|
184
|
+
export declare class AdaptivePlaywrightCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<AdaptivePlaywrightCrawlerContext['request']>>, StatisticStateExtension extends AdaptivePlaywrightCrawlerStatisticState = AdaptivePlaywrightCrawlerStatisticState> extends BasicCrawler<AdaptivePlaywrightCrawlerContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
185
|
+
#private;
|
|
186
|
+
constructor(options?: AdaptivePlaywrightCrawlerOptions<ContextExtension, ExtendedContext, Routes, StatisticStateExtension>);
|
|
187
|
+
protected init(): Promise<void>;
|
|
188
|
+
private adaptCheerioContext;
|
|
189
|
+
private adaptPlaywrightContext;
|
|
190
|
+
/**
|
|
191
|
+
* Runs one request handler attempt inside its own {@link StorageTransaction}, wrapping the inner
|
|
192
|
+
* (static or browser) context pipeline. The transaction is pushed to `transactions` *at creation
|
|
193
|
+
* time, before the `try`* - the `ok: false` branch of the returned {@link Result} carries no
|
|
194
|
+
* result, and failed attempts are routine here. The caller owns the outcome and disposal.
|
|
177
195
|
*/
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
protected _runRequestHandler(crawlingContext: PlaywrightCrawlingContext): Promise<void>;
|
|
182
|
-
protected commitResult(crawlingContext: PlaywrightCrawlingContext, { calls, keyValueStoreChanges }: RequestHandlerResult): Promise<void>;
|
|
183
|
-
protected allowStorageAccess<R, TArgs extends any[]>(func: (...args: TArgs) => Promise<R>): (...args: TArgs) => Promise<R>;
|
|
184
|
-
protected runRequestHandlerInBrowser(crawlingContext: PlaywrightCrawlingContext): Promise<{
|
|
185
|
-
result: Result<RequestHandlerResult>;
|
|
186
|
-
initialStateCopy?: Record<string, unknown>;
|
|
187
|
-
}>;
|
|
188
|
-
protected runRequestHandlerWithPlainHTTP(crawlingContext: PlaywrightCrawlingContext, oldStateCopy?: Dictionary): Promise<Result<RequestHandlerResult>>;
|
|
196
|
+
private crawlOne;
|
|
197
|
+
protected runRequestHandler(crawlingContext: CrawlingContext): Promise<void>;
|
|
198
|
+
private enqueueLinks;
|
|
189
199
|
private createLogProxy;
|
|
200
|
+
/**
|
|
201
|
+
* Number of rendering type detections that have not settled yet, including results the predictor is
|
|
202
|
+
* still persisting.
|
|
203
|
+
*/
|
|
204
|
+
get inFlightRenderingTypeDetectionCount(): number;
|
|
205
|
+
/**
|
|
206
|
+
* Waits for in-flight rendering type detections to settle, bounded by `timeoutMillis` (defaults to the
|
|
207
|
+
* internal timeout).
|
|
208
|
+
*/
|
|
209
|
+
drainRenderingDetections({ timeoutMillis }?: {
|
|
210
|
+
timeoutMillis?: number;
|
|
211
|
+
}): Promise<void>;
|
|
212
|
+
/**
|
|
213
|
+
* Stops the crawler immediately, but not before rendering type detections already under way (and results
|
|
214
|
+
* the predictor is still persisting) have settled - see
|
|
215
|
+
* {@link AdaptivePlaywrightCrawler.drainRenderingDetections|`drainRenderingDetections()`}. Requests
|
|
216
|
+
* that are still running are not waited for, unlike {@link BasicCrawler.stop|`stop()`}.
|
|
217
|
+
*/
|
|
218
|
+
teardown(): Promise<void>;
|
|
219
|
+
destroy(): Promise<void>;
|
|
190
220
|
}
|
|
191
|
-
export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext,
|
|
221
|
+
export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
222
|
+
export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
223
|
+
export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
224
|
+
/**
|
|
225
|
+
* An opt-in {@link AdaptivePlaywrightCrawlerOptions.resultComparator|`resultComparator`} that considers two
|
|
226
|
+
* request handler results equal only if *all* of their observable effects match - the pushed dataset items, the
|
|
227
|
+
* enqueued requests, and the key-value store changes. This is stricter than the default comparator, which only
|
|
228
|
+
* compares dataset items.
|
|
229
|
+
*
|
|
230
|
+
* **Beware:** enqueued URLs are compared exactly. The same page rendered in a browser and via plain HTTP often
|
|
231
|
+
* yields links that differ only in tracking query parameters, for example:
|
|
232
|
+
* - `https://sdk.apify.com/docs/guides/getting-started`
|
|
233
|
+
* - `https://sdk.apify.com/docs/guides/getting-started?__hsfp=1136113150&__hssc=7591405.1.173549427712`
|
|
234
|
+
*
|
|
235
|
+
* Such links are treated as *different*, which will make the crawler favor browser rendering for those pages.
|
|
236
|
+
*
|
|
237
|
+
* **Example usage:**
|
|
238
|
+
* ```ts
|
|
239
|
+
* const crawler = new AdaptivePlaywrightCrawler({
|
|
240
|
+
* resultComparator: fullResultComparator,
|
|
241
|
+
* async requestHandler({ pushData, enqueueLinks }) {
|
|
242
|
+
* // ...
|
|
243
|
+
* },
|
|
244
|
+
* });
|
|
245
|
+
* ```
|
|
246
|
+
*/
|
|
247
|
+
export declare function fullResultComparator(resultA: StorageTransactionView, resultB: StorageTransactionView): boolean;
|
|
192
248
|
export {};
|
|
193
|
-
//# sourceMappingURL=adaptive-playwright-crawler.d.ts.map
|