@crawlee/jsdom 4.0.0-beta.11 → 4.0.0-beta.110

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,19 +1,23 @@
1
1
  <h1 align="center">
2
2
  <a href="https://crawlee.dev">
3
3
  <picture>
4
- <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true">
5
- <img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500">
4
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true" />
5
+ <img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500" />
6
6
  </picture>
7
7
  </a>
8
- <br>
8
+ <br />
9
9
  <small>A web scraping and browser automation library</small>
10
10
  </h1>
11
11
 
12
- <p align=center>
13
- <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" style="max-width: 100%;"></a>
14
- <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" style="max-width: 100%;"></a>
15
- <a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" style="max-width: 100%;"></a>
16
- <a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" style="max-width: 100%;"></a>
12
+ <p align="center">
13
+ <a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" width="250" height="55"/></a>
14
+ </p>
15
+
16
+ <p align="center">
17
+ <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" /></a>
18
+ <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" /></a>
19
+ <a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" /></a>
20
+ <a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" /></a>
17
21
  </p>
18
22
 
19
23
  Crawlee covers your crawling and scraping end-to-end and **helps you build reliable scrapers. Fast.**
@@ -24,7 +28,7 @@ Crawlee is available as the [`crawlee`](https://www.npmjs.com/package/crawlee) N
24
28
 
25
29
  > 👉 **View full documentation, guides and examples on the [Crawlee project website](https://crawlee.dev)** 👈
26
30
 
27
- > Crawlee for Python is open for early adopters. 🐍 [👉 Checkout the source code 👈](https://github.com/apify/crawlee-python).
31
+ > Do you prefer 🐍 Python instead of JavaScript? [👉 Checkout Crawlee for Python 👈](https://github.com/apify/crawlee-python).
28
32
 
29
33
  ## Installation
30
34
 
@@ -85,7 +89,7 @@ By default, Crawlee stores data to `./storage` in the current working directory.
85
89
  We provide automated beta builds for every merged code change in Crawlee. You can find them in the npm [list of releases](https://www.npmjs.com/package/crawlee?activeTab=versions). If you want to test new features or bug fixes before we release them, feel free to install a beta build like this:
86
90
 
87
91
  ```bash
88
- npm install crawlee@3.12.3-beta.13
92
+ npm install crawlee@next
89
93
  ```
90
94
 
91
95
  If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need to specify dependency overrides in your `package.json` file so that you don't end up with multiple versions of Crawlee installed:
@@ -94,9 +98,9 @@ If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need
94
98
  {
95
99
  "overrides": {
96
100
  "apify": {
97
- "@crawlee/core": "3.12.3-beta.13",
98
- "@crawlee/types": "3.12.3-beta.13",
99
- "@crawlee/utils": "3.12.3-beta.13"
101
+ "@crawlee/core": "$crawlee",
102
+ "@crawlee/types": "$crawlee",
103
+ "@crawlee/utils": "$crawlee"
100
104
  }
101
105
  }
102
106
  }
package/index.d.ts CHANGED
@@ -1,3 +1,2 @@
1
1
  export * from '@crawlee/http';
2
2
  export * from './internals/jsdom-crawler.js';
3
- //# sourceMappingURL=index.d.ts.map
package/index.js CHANGED
@@ -1,3 +1,2 @@
1
1
  export * from '@crawlee/http';
2
2
  export * from './internals/jsdom-crawler.js';
3
- //# sourceMappingURL=index.js.map
@@ -1,13 +1,15 @@
1
- import type { Configuration, EnqueueLinksOptions, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, RequestHandler, RequestProvider, RouterRoutes, SkippedRequestCallback } from '@crawlee/http';
1
+ import type { BasicCrawlingContext, CrawlingContext, EnqueueLinksOptions, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, IRequestManager, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas, SkippedRequestCallback } from '@crawlee/http';
2
2
  import { HttpCrawler } from '@crawlee/http';
3
3
  import type { Dictionary } from '@crawlee/types';
4
4
  import { type CheerioRoot, type RobotsTxtFile } from '@crawlee/utils';
5
5
  import type { DOMWindow } from 'jsdom';
6
6
  import { VirtualConsole } from 'jsdom';
7
7
  export type JSDOMErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
8
- JSONData extends Dictionary = any> = ErrorHandler<JSDOMCrawlingContext<UserData, JSONData>>;
9
- export interface JSDOMCrawlerOptions<ExtendedContext extends JSDOMCrawlingContext = JSDOMCrawlingContext, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
10
- JSONData extends Dictionary = any> extends HttpCrawlerOptions<JSDOMCrawlingContext<UserData, JSONData>, ExtendedContext> {
8
+ JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
9
+ ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, JSDOMCrawlingContext<UserData, JSONData> & ContextExtension>;
10
+ export interface JSDOMCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends JSDOMCrawlingContext = JSDOMCrawlingContext & ContextExtension, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
11
+ JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
12
+ Routes extends Record<keyof Routes, Dictionary> = Record<string, UserData>> extends HttpCrawlerOptions<JSDOMCrawlingContext<UserData, JSONData>, ContextExtension, ExtendedContext, Routes> {
11
13
  /**
12
14
  * Download and run scripts.
13
15
  */
@@ -54,7 +56,8 @@ JSONData extends Dictionary = any> extends InternalHttpCrawlingContext<UserData,
54
56
  }
55
57
  export type JSDOMRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
56
58
  JSONData extends Dictionary = any> = RequestHandler<JSDOMCrawlingContext<UserData, JSONData>>;
57
- export declare class JSDOMCrawler<ExtendedContext extends JSDOMCrawlingContext = JSDOMCrawlingContext> extends HttpCrawler<JSDOMCrawlingContext, ExtendedContext> {
59
+ export declare class JSDOMCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends JSDOMCrawlingContext = JSDOMCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<JSDOMCrawlingContext['request']>>> extends HttpCrawler<JSDOMCrawlingContext, ContextExtension, ExtendedContext, Routes> {
60
+ #private;
58
61
  protected static optionsShape: {
59
62
  // @ts-ignore optional peer dependency or compatibility with es2022
60
63
  runScripts: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
@@ -71,13 +74,7 @@ export declare class JSDOMCrawler<ExtendedContext extends JSDOMCrawlingContext =
71
74
  // @ts-ignore optional peer dependency or compatibility with es2022
72
75
  forceResponseEncoding: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
73
76
  // @ts-ignore optional peer dependency or compatibility with es2022
74
- proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
75
- // @ts-ignore optional peer dependency or compatibility with es2022
76
- persistCookiesPerSession: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
77
- // @ts-ignore optional peer dependency or compatibility with es2022
78
- additionalHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
79
- // @ts-ignore optional peer dependency or compatibility with es2022
80
- ignoreHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
77
+ saveResponseCookies: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
81
78
  // @ts-ignore optional peer dependency or compatibility with es2022
82
79
  preNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
83
80
  // @ts-ignore optional peer dependency or compatibility with es2022
@@ -102,28 +99,46 @@ export declare class JSDOMCrawler<ExtendedContext extends JSDOMCrawlingContext =
102
99
  maxRequestRetries: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
103
100
  // @ts-ignore optional peer dependency or compatibility with es2022
104
101
  sameDomainDelaySecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
105
- // @ts-ignore optional peer dependency or compatibility with es2022
106
- maxSessionRotations: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
107
102
  // @ts-ignore optional peer dependency or compatibility with es2022
108
103
  maxRequestsPerCrawl: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
109
104
  // @ts-ignore optional peer dependency or compatibility with es2022
110
- autoscaledPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
105
+ maxCrawlDepth: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
111
106
  // @ts-ignore optional peer dependency or compatibility with es2022
112
- sessionPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
107
+ taskLoopOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
113
108
  // @ts-ignore optional peer dependency or compatibility with es2022
114
- useSessionPool: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
109
+ concurrencySystem: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
110
+ // @ts-ignore optional peer dependency or compatibility with es2022
111
+ sessionPool: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
112
+ // @ts-ignore optional peer dependency or compatibility with es2022
113
+ proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
115
114
  // @ts-ignore optional peer dependency or compatibility with es2022
116
115
  statusMessageLoggingInterval: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
117
116
  // @ts-ignore optional peer dependency or compatibility with es2022
118
117
  statusMessageCallback: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
118
+ // @ts-ignore optional peer dependency or compatibility with es2022
119
+ additionalHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
120
+ // @ts-ignore optional peer dependency or compatibility with es2022
121
+ ignoreHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
122
+ // @ts-ignore optional peer dependency or compatibility with es2022
123
+ blockedStatusCodes: import("ow").ArrayPredicate<number>;
119
124
  // @ts-ignore optional peer dependency or compatibility with es2022
120
125
  retryOnBlocked: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
121
126
  // @ts-ignore optional peer dependency or compatibility with es2022
122
- respectRobotsTxtFile: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
127
+ respectRobotsTxtFile: import("ow").AnyPredicate<boolean | object>;
128
+ // @ts-ignore optional peer dependency or compatibility with es2022
129
+ transactionalStorage: import("ow").BasePredicate<boolean | Partial<import("@crawlee/http").StorageWritePolicy> | undefined>;
123
130
  // @ts-ignore optional peer dependency or compatibility with es2022
124
131
  onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
125
132
  // @ts-ignore optional peer dependency or compatibility with es2022
126
133
  httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
134
+ // @ts-ignore optional peer dependency or compatibility with es2022
135
+ configuration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
136
+ // @ts-ignore optional peer dependency or compatibility with es2022
137
+ storageBackend: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
138
+ // @ts-ignore optional peer dependency or compatibility with es2022
139
+ eventManager: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
140
+ // @ts-ignore optional peer dependency or compatibility with es2022
141
+ logger: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
127
142
  // @ts-ignore optional peer dependency or compatibility with es2022
128
143
  minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
129
144
  // @ts-ignore optional peer dependency or compatibility with es2022
@@ -133,16 +148,22 @@ export declare class JSDOMCrawler<ExtendedContext extends JSDOMCrawlingContext =
133
148
  // @ts-ignore optional peer dependency or compatibility with es2022
134
149
  keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
135
150
  // @ts-ignore optional peer dependency or compatibility with es2022
136
- log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
137
- // @ts-ignore optional peer dependency or compatibility with es2022
138
- experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
151
+ statistics: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
139
152
  // @ts-ignore optional peer dependency or compatibility with es2022
140
- statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
153
+ id: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
141
154
  };
142
- protected runScripts: boolean;
143
- protected hideInternalConsole: boolean;
144
- protected virtualConsole: VirtualConsole | null;
145
- constructor(options?: JSDOMCrawlerOptions<ExtendedContext>, config?: Configuration);
155
+ constructor(options?: JSDOMCrawlerOptions<ContextExtension, ExtendedContext, any, any, Routes>);
156
+ // @ts-ignore optional peer dependency or compatibility with es2022
157
+ protected buildContextPipeline(): import("@crawlee/http").ContextPipeline<CrawlingContext<Dictionary>, InternalHttpCrawlingContext<any, any> & {
158
+ readonly window: DOMWindow;
159
+ readonly body: string;
160
+ readonly document: Document;
161
+ } & {
162
+ enqueueLinks: (enqueueOptions?: EnqueueLinksOptions) => Promise<unknown>;
163
+ waitForSelector(selector: string, timeoutMs?: number): Promise<void>;
164
+ // @ts-ignore optional peer dependency or compatibility with es2022
165
+ parseWithCheerio(selector?: string, _timeoutMs?: number): Promise<import("cheerio").CheerioAPI>;
166
+ }>;
146
167
  /**
147
168
  * Returns the currently used `VirtualConsole` instance. Can be used to listen for the JSDOM's internal console messages.
148
169
  *
@@ -165,15 +186,21 @@ export declare class JSDOMCrawler<ExtendedContext extends JSDOMCrawlingContext =
165
186
  interface EnqueueLinksInternalOptions {
166
187
  options?: EnqueueLinksOptions;
167
188
  window: DOMWindow | null;
168
- requestQueue: RequestProvider;
189
+ requestManager: IRequestManager;
169
190
  robotsTxtFile?: RobotsTxtFile;
170
191
  onSkippedRequest?: SkippedRequestCallback;
171
192
  originalRequestUrl: string;
172
193
  finalRequestUrl?: string;
173
194
  }
195
+ interface BoundEnqueueLinksInternalOptions {
196
+ enqueueLinks: BasicCrawlingContext['enqueueLinks'];
197
+ options?: EnqueueLinksOptions;
198
+ window: DOMWindow | null;
199
+ originalRequestUrl: string;
200
+ finalRequestUrl?: string;
201
+ }
174
202
  /** @internal */
175
- // @ts-ignore optional peer dependency or compatibility with es2022
176
- export declare function domCrawlerEnqueueLinks({ options, window, requestQueue, robotsTxtFile, onSkippedRequest, originalRequestUrl, finalRequestUrl, }: EnqueueLinksInternalOptions): Promise<import("@crawlee/types").BatchAddRequestsResult>;
203
+ export declare function domCrawlerEnqueueLinks(options: EnqueueLinksInternalOptions | BoundEnqueueLinksInternalOptions): Promise<unknown>;
177
204
  /**
178
205
  * Creates new {@link Router} instance that works based on request labels.
179
206
  * This instance can then serve as a `requestHandler` of your {@link JSDOMCrawler}.
@@ -198,7 +225,7 @@ export declare function domCrawlerEnqueueLinks({ options, window, requestQueue,
198
225
  * await crawler.run();
199
226
  * ```
200
227
  */
201
- // @ts-ignore optional peer dependency or compatibility with es2022
202
- export declare function createJSDOMRouter<Context extends JSDOMCrawlingContext = JSDOMCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData>): import("@crawlee/http").RouterHandler<Context>;
228
+ export declare function createJSDOMRouter<Context extends JSDOMCrawlingContext = JSDOMCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
229
+ export declare function createJSDOMRouter<Context extends JSDOMCrawlingContext = JSDOMCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
230
+ export declare function createJSDOMRouter<Context extends JSDOMCrawlingContext = JSDOMCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
203
231
  export {};
204
- //# sourceMappingURL=jsdom-crawler.d.ts.map
@@ -1,6 +1,5 @@
1
- import { enqueueLinks, HttpCrawler, resolveBaseUrlForEnqueueLinksFiltering, Router, tryAbsoluteURL, } from '@crawlee/http';
1
+ import { enqueueLinks, HttpCrawler, NavigationSkippedError, resolveBaseUrlForEnqueueLinksFiltering, Router, tryAbsoluteURL, } from '@crawlee/http';
2
2
  import { sleep } from '@crawlee/utils';
3
- import * as cheerio from 'cheerio';
4
3
  import { JSDOM, ResourceLoader, VirtualConsole } from 'jsdom';
5
4
  import ow from 'ow';
6
5
  import { addTimeoutToPromise } from '@apify/timeout';
@@ -26,38 +25,40 @@ import { addTimeoutToPromise } from '@apify/timeout';
26
25
  * and then invokes the user-provided {@link JSDOMCrawlerOptions.requestHandler} to extract page data
27
26
  * using the `window` object.
28
27
  *
29
- * The source URLs are represented using {@link Request} objects that are fed from
30
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link JSDOMCrawlerOptions.requestList}
31
- * or {@link JSDOMCrawlerOptions.requestQueue} constructor options, respectively.
28
+ * The source URLs are represented using {@link Request} objects that are fed from the
29
+ * {@link IRequestManager|request manager} provided via the {@link JSDOMCrawlerOptions.requestManager|`requestManager`}
30
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
31
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
32
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
33
+ * result as `requestManager`.
32
34
  *
33
- * If both {@link JSDOMCrawlerOptions.requestList} and {@link JSDOMCrawlerOptions.requestQueue} are used,
34
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
35
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
35
+ * > The {@link JSDOMCrawlerOptions.requestList|`requestList`} and {@link JSDOMCrawlerOptions.requestQueue|`requestQueue`}
36
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
36
37
  *
37
38
  * The crawler finishes when there are no more {@link Request} objects to crawl.
38
39
  *
39
- * We can use the `preNavigationHooks` to adjust `gotOptions`:
40
+ * We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
40
41
  *
41
42
  * ```
42
43
  * preNavigationHooks: [
43
- * (crawlingContext, gotOptions) => {
44
+ * (crawlingContext) => {
44
45
  * // ...
45
46
  * },
46
47
  * ]
47
48
  * ```
48
49
  *
49
- * By default, `JSDOMCrawler` only processes web pages with the `text/html`
50
- * and `application/xhtml+xml` MIME content types (as reported by the `Content-Type` HTTP header),
50
+ * By default, `JSDOMCrawler` only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
51
+ * and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
51
52
  * and skips pages with other content types. If you want the crawler to process other content types,
52
53
  * use the {@link JSDOMCrawlerOptions.additionalMimeTypes} constructor option.
53
54
  * Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
54
55
  * For more details, see {@link JSDOMCrawlerOptions.requestHandler}.
55
56
  *
56
- * New requests are only dispatched when there is enough free CPU and memory available,
57
- * using the functionality provided by the {@link AutoscaledPool} class.
58
- * All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions`
59
- * parameter of the `CheerioCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
60
- * {@link AutoscaledPool} options are available directly in the `CheerioCrawler` constructor.
57
+ * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
58
+ * {@link ConcurrencySystem}.
59
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
60
+ * `JSDOMCrawler` constructor, or, for finer control, by injecting a pre-configured
61
+ * {@link ConcurrencySystem|`concurrencySystem`}.
61
62
  *
62
63
  * **Example usage:**
63
64
  *
@@ -88,25 +89,29 @@ export class JSDOMCrawler extends HttpCrawler {
88
89
  runScripts: ow.optional.boolean,
89
90
  hideInternalConsole: ow.optional.boolean,
90
91
  };
91
- runScripts;
92
- hideInternalConsole;
93
- virtualConsole = null;
94
- constructor(options = {}, config) {
95
- const { runScripts = false, hideInternalConsole = false, ...httpOptions } = options;
92
+ #runScripts;
93
+ #hideInternalConsole;
94
+ #virtualConsole = null;
95
+ constructor(options = {}) {
96
+ const { runScripts = false, hideInternalConsole = false, contextPipelineBuilder, ...httpOptions } = options;
96
97
  super({
97
98
  ...httpOptions,
98
- contextPipelineBuilder: () => this.buildContextPipeline()
99
- .compose({
100
- action: async (context) => await this.parseContent(context),
101
- cleanup: async (context) => {
102
- this.getVirtualConsole().off('jsdomError', this.jsdomErrorHandler);
103
- context.window?.close();
104
- },
105
- })
106
- .compose({ action: async (context) => await this.addHelpers(context) }),
107
- }, config);
108
- this.runScripts = runScripts;
109
- this.hideInternalConsole = hideInternalConsole;
99
+ contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()),
100
+ });
101
+ this.#runScripts = runScripts;
102
+ this.#hideInternalConsole = hideInternalConsole;
103
+ }
104
+ buildContextPipeline() {
105
+ return super
106
+ .buildContextPipeline()
107
+ .compose({
108
+ action: async (context) => await this.parseContent(context),
109
+ cleanup: async (context) => {
110
+ this.getVirtualConsole().off('jsdomError', this.jsdomErrorHandler);
111
+ context.window?.close();
112
+ },
113
+ })
114
+ .compose({ action: async (context) => await this.addHelpers(context) });
110
115
  }
111
116
  /**
112
117
  * Returns the currently used `VirtualConsole` instance. Can be used to listen for the JSDOM's internal console messages.
@@ -123,86 +128,108 @@ export class JSDOMCrawler extends HttpCrawler {
123
128
  * ```
124
129
  */
125
130
  getVirtualConsole() {
126
- if (this.virtualConsole) {
127
- return this.virtualConsole;
131
+ if (this.#virtualConsole) {
132
+ return this.#virtualConsole;
128
133
  }
129
- this.virtualConsole = new VirtualConsole();
130
- if (!this.hideInternalConsole) {
131
- this.virtualConsole.sendTo(console, { omitJSDOMErrors: true });
134
+ this.#virtualConsole = new VirtualConsole();
135
+ if (!this.#hideInternalConsole) {
136
+ this.#virtualConsole.sendTo(console, { omitJSDOMErrors: true });
132
137
  }
133
- this.virtualConsole.on('jsdomError', this.jsdomErrorHandler);
134
- return this.virtualConsole;
138
+ this.#virtualConsole.on('jsdomError', this.jsdomErrorHandler);
139
+ return this.#virtualConsole;
135
140
  }
136
- jsdomErrorHandler = (error) => this.log.debug('JSDOM error from console', error);
141
+ jsdomErrorHandler = (error) => this.log.debug('JSDOM error from console', { error });
137
142
  async parseContent(crawlingContext) {
138
- const isXml = crawlingContext.contentType.type.includes('xml');
139
- // TODO handle non-string
140
- const { window } = new JSDOM(crawlingContext.body.toString(), {
141
- url: crawlingContext.response.url,
142
- contentType: isXml ? 'text/xml' : 'text/html',
143
- runScripts: this.runScripts ? 'dangerously' : undefined,
144
- resources,
145
- virtualConsole: this.getVirtualConsole(),
146
- pretendToBeVisual: true,
147
- });
148
- // add some stubs in place of missing API so processing won't fail
149
- Object.defineProperty(window, 'matchMedia', {
150
- writable: true,
151
- value: (query) => ({
152
- matches: false,
153
- media: query,
154
- onchange: null,
155
- addListener: () => { },
156
- removeListener: () => { },
157
- addEventListener: () => { },
158
- removeEventListener: () => { },
159
- dispatchEvent: () => { },
160
- }),
161
- });
162
- window.document.createRange = () => {
163
- const range = new window.Range();
164
- range.getBoundingClientRect = () => ({});
165
- range.getClientRects = () => ({ item: () => null, length: 0 });
166
- return range;
167
- };
168
- if (this.runScripts) {
169
- try {
170
- await addTimeoutToPromise(async () => {
171
- return new Promise((resolve) => {
172
- window.addEventListener('load', () => {
173
- resolve();
174
- }, false);
175
- }).catch();
176
- }, 10_000, 'Window.load event not fired after 10 seconds.').catch();
143
+ try {
144
+ const isXml = crawlingContext.contentType.type.includes('xml');
145
+ // TODO handle non-string
146
+ const { window } = new JSDOM(crawlingContext.body.toString(), {
147
+ url: crawlingContext.response.url,
148
+ contentType: isXml ? 'text/xml' : 'text/html',
149
+ runScripts: this.#runScripts ? 'dangerously' : undefined,
150
+ resources,
151
+ virtualConsole: this.getVirtualConsole(),
152
+ pretendToBeVisual: true,
153
+ });
154
+ // add some stubs in place of missing API so processing won't fail
155
+ Object.defineProperty(window, 'matchMedia', {
156
+ writable: true,
157
+ value: (query) => ({
158
+ matches: false,
159
+ media: query,
160
+ onchange: null,
161
+ addListener: () => { },
162
+ removeListener: () => { },
163
+ addEventListener: () => { },
164
+ removeEventListener: () => { },
165
+ dispatchEvent: () => { },
166
+ }),
167
+ });
168
+ window.document.createRange = () => {
169
+ const range = new window.Range();
170
+ range.getBoundingClientRect = () => ({});
171
+ range.getClientRects = () => ({ item: () => null, length: 0 });
172
+ return range;
173
+ };
174
+ if (this.#runScripts) {
175
+ try {
176
+ await addTimeoutToPromise(async () => {
177
+ return new Promise((resolve) => {
178
+ window.addEventListener('load', () => {
179
+ resolve();
180
+ }, false);
181
+ }).catch();
182
+ }, 10_000, 'Window.load event not fired after 10 seconds.').catch();
183
+ }
184
+ catch (e) {
185
+ this.log.debug(e.message);
186
+ }
177
187
  }
178
- catch (e) {
179
- this.log.debug(e.message);
188
+ return {
189
+ window,
190
+ get body() {
191
+ return window.document.documentElement.outerHTML;
192
+ },
193
+ get document() {
194
+ return window.document;
195
+ },
196
+ };
197
+ }
198
+ catch (err) {
199
+ if (err instanceof NavigationSkippedError) {
200
+ return {
201
+ get window() {
202
+ throw new NavigationSkippedError('The `window` property is not available - `skipNavigation` was used', { cause: err });
203
+ },
204
+ get body() {
205
+ throw new NavigationSkippedError('The `body` property is not available - `skipNavigation` was used', { cause: err });
206
+ },
207
+ get document() {
208
+ throw new NavigationSkippedError('The `document` property is not available - `skipNavigation` was used', { cause: err });
209
+ },
210
+ };
180
211
  }
212
+ throw err;
181
213
  }
182
- return {
183
- window,
184
- get body() {
185
- return window.document.documentElement.outerHTML;
186
- },
187
- get document() {
188
- return window.document;
189
- },
190
- };
191
214
  }
192
215
  async addHelpers(crawlingContext) {
193
216
  return {
194
217
  enqueueLinks: async (enqueueOptions) => {
195
218
  return domCrawlerEnqueueLinks({
196
- options: enqueueOptions,
219
+ options: {
220
+ ...enqueueOptions,
221
+ limit: await this.calculateEnqueuedRequestLimit(enqueueOptions?.limit),
222
+ },
197
223
  window: crawlingContext.window,
198
- requestQueue: await this.getRequestQueue(),
224
+ requestManager: await this.getRequestManager(),
199
225
  robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
200
- onSkippedRequest: this.onSkippedRequest,
226
+ onSkippedRequest: this.handleSkippedRequest,
201
227
  originalRequestUrl: crawlingContext.request.url,
202
228
  finalRequestUrl: crawlingContext.request.loadedUrl,
203
229
  });
204
230
  },
205
231
  async waitForSelector(selector, timeoutMs = 5_000) {
232
+ const cheerio = await import('cheerio');
206
233
  const $ = cheerio.load(crawlingContext.body);
207
234
  if ($(selector).get().length === 0) {
208
235
  if (timeoutMs) {
@@ -214,6 +241,7 @@ export class JSDOMCrawler extends HttpCrawler {
214
241
  }
215
242
  },
216
243
  async parseWithCheerio(selector, _timeoutMs = 5_000) {
244
+ const cheerio = await import('cheerio');
217
245
  const $ = cheerio.load(crawlingContext.body);
218
246
  if (selector && $(selector).get().length === 0) {
219
247
  throw new Error(`Selector '${selector}' not found.`);
@@ -224,24 +252,36 @@ export class JSDOMCrawler extends HttpCrawler {
224
252
  }
225
253
  }
226
254
  /** @internal */
227
- export async function domCrawlerEnqueueLinks({ options, window, requestQueue, robotsTxtFile, onSkippedRequest, originalRequestUrl, finalRequestUrl, }) {
255
+ function containsEnqueueLinks(options) {
256
+ return !!options.enqueueLinks;
257
+ }
258
+ /** @internal */
259
+ export async function domCrawlerEnqueueLinks(options) {
260
+ const { options: enqueueLinksOptions, window, originalRequestUrl, finalRequestUrl } = options;
228
261
  if (!window) {
229
262
  throw new Error('Cannot enqueue links because the JSDOM is not available.');
230
263
  }
231
264
  const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
232
- enqueueStrategy: options?.strategy,
265
+ enqueueStrategy: enqueueLinksOptions?.strategy,
233
266
  finalRequestUrl,
234
267
  originalRequestUrl,
235
- userProvidedBaseUrl: options?.baseUrl,
268
+ userProvidedBaseUrl: enqueueLinksOptions?.baseUrl,
236
269
  });
237
- const urls = extractUrlsFromWindow(window, options?.selector ?? 'a', options?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
270
+ const urls = extractUrlsFromWindow(window, enqueueLinksOptions?.selector ?? 'a', enqueueLinksOptions?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
271
+ if (containsEnqueueLinks(options)) {
272
+ return options.enqueueLinks({
273
+ urls,
274
+ baseUrl,
275
+ ...enqueueLinksOptions,
276
+ });
277
+ }
238
278
  return enqueueLinks({
239
- requestQueue,
240
- robotsTxtFile,
241
- onSkippedRequest,
279
+ requestManager: options.requestManager,
280
+ robotsTxtFile: options.robotsTxtFile,
281
+ onSkippedRequest: options.onSkippedRequest,
242
282
  urls,
243
283
  baseUrl,
244
- ...options,
284
+ ...enqueueLinksOptions,
245
285
  });
246
286
  }
247
287
  /**
@@ -260,31 +300,6 @@ function extractUrlsFromWindow(window, selector, baseUrl) {
260
300
  })
261
301
  .filter((href) => href !== undefined && href !== '');
262
302
  }
263
- /**
264
- * Creates new {@link Router} instance that works based on request labels.
265
- * This instance can then serve as a `requestHandler` of your {@link JSDOMCrawler}.
266
- * Defaults to the {@link JSDOMCrawlingContext}.
267
- *
268
- * > Serves as a shortcut for using `Router.create<JSDOMCrawlingContext>()`.
269
- *
270
- * ```ts
271
- * import { JSDOMCrawler, createJSDOMRouter } from 'crawlee';
272
- *
273
- * const router = createJSDOMRouter();
274
- * router.addHandler('label-a', async (ctx) => {
275
- * ctx.log.info('...');
276
- * });
277
- * router.addDefaultHandler(async (ctx) => {
278
- * ctx.log.info('...');
279
- * });
280
- *
281
- * const crawler = new JSDOMCrawler({
282
- * requestHandler: router,
283
- * });
284
- * await crawler.run();
285
- * ```
286
- */
287
- export function createJSDOMRouter(routes) {
288
- return Router.create(routes);
303
+ export function createJSDOMRouter(routesOrSchemas) {
304
+ return Router.create(routesOrSchemas);
289
305
  }
290
- //# sourceMappingURL=jsdom-crawler.js.map