@crawlee/cheerio 4.0.0-beta.19 → 4.0.0-beta.190
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +1 -1
- package/index.js +0 -1
- package/internals/cheerio-crawler.d.ts +29 -88
- package/internals/cheerio-crawler.js +22 -129
- package/internals/cheerio-parser.d.ts +11 -0
- package/internals/cheerio-parser.js +27 -0
- package/package.json +6 -6
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/cheerio-crawler.d.ts.map +0 -1
- package/internals/cheerio-crawler.js.map +0 -1
package/README.md
CHANGED
|
@@ -1,23 +1,23 @@
|
|
|
1
1
|
<h1 align="center">
|
|
2
2
|
<a href="https://crawlee.dev">
|
|
3
3
|
<picture>
|
|
4
|
-
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true"
|
|
5
|
-
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500"
|
|
4
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true" />
|
|
5
|
+
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500" />
|
|
6
6
|
</picture>
|
|
7
7
|
</a>
|
|
8
|
-
<br
|
|
8
|
+
<br />
|
|
9
9
|
<small>A web scraping and browser automation library</small>
|
|
10
10
|
</h1>
|
|
11
11
|
|
|
12
|
-
<p align=center>
|
|
13
|
-
<a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift"
|
|
12
|
+
<p align="center">
|
|
13
|
+
<a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" width="250" height="55"/></a>
|
|
14
14
|
</p>
|
|
15
15
|
|
|
16
|
-
<p align=center>
|
|
17
|
-
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg"
|
|
18
|
-
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg"
|
|
19
|
-
<a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord"
|
|
20
|
-
<a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status"
|
|
16
|
+
<p align="center">
|
|
17
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" /></a>
|
|
18
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" /></a>
|
|
19
|
+
<a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" /></a>
|
|
20
|
+
<a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" /></a>
|
|
21
21
|
</p>
|
|
22
22
|
|
|
23
23
|
Crawlee covers your crawling and scraping end-to-end and **helps you build reliable scrapers. Fast.**
|
|
@@ -89,7 +89,7 @@ By default, Crawlee stores data to `./storage` in the current working directory.
|
|
|
89
89
|
We provide automated beta builds for every merged code change in Crawlee. You can find them in the npm [list of releases](https://www.npmjs.com/package/crawlee?activeTab=versions). If you want to test new features or bug fixes before we release them, feel free to install a beta build like this:
|
|
90
90
|
|
|
91
91
|
```bash
|
|
92
|
-
npm install crawlee@
|
|
92
|
+
npm install crawlee@next
|
|
93
93
|
```
|
|
94
94
|
|
|
95
95
|
If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need to specify dependency overrides in your `package.json` file so that you don't end up with multiple versions of Crawlee installed:
|
|
@@ -98,9 +98,9 @@ If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need
|
|
|
98
98
|
{
|
|
99
99
|
"overrides": {
|
|
100
100
|
"apify": {
|
|
101
|
-
"@crawlee/core": "
|
|
102
|
-
"@crawlee/types": "
|
|
103
|
-
"@crawlee/utils": "
|
|
101
|
+
"@crawlee/core": "$crawlee",
|
|
102
|
+
"@crawlee/types": "$crawlee",
|
|
103
|
+
"@crawlee/utils": "$crawlee"
|
|
104
104
|
}
|
|
105
105
|
}
|
|
106
106
|
}
|
package/index.d.ts
CHANGED
package/index.js
CHANGED
|
@@ -1,58 +1,18 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
import {
|
|
3
|
-
import type {
|
|
4
|
-
import
|
|
5
|
-
import * as cheerio from 'cheerio';
|
|
1
|
+
import type { CrawlingContext, DOMCrawlingContext, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/http';
|
|
2
|
+
import { DOMCrawler } from '@crawlee/http';
|
|
3
|
+
import type { Dictionary } from '@crawlee/types';
|
|
4
|
+
import type { CheerioParseResult } from './cheerio-parser.js';
|
|
6
5
|
export type CheerioErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
7
|
-
JSONData extends Dictionary = any
|
|
6
|
+
JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
7
|
+
ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, CheerioCrawlingContext<UserData, JSONData> & ContextExtension>;
|
|
8
8
|
export interface CheerioCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
9
|
-
JSONData extends Dictionary = any
|
|
9
|
+
JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
10
|
+
Routes extends Record<keyof Routes, Dictionary> = Record<string, UserData>, StatisticStateExtension extends object = {}> extends HttpCrawlerOptions<CheerioCrawlingContext<UserData, JSONData>, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
10
11
|
}
|
|
11
12
|
export type CheerioHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
12
13
|
JSONData extends Dictionary = any> = InternalHttpHook<CheerioCrawlingContext<UserData, JSONData>>;
|
|
13
14
|
export interface CheerioCrawlingContext<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
14
|
-
JSONData extends Dictionary = any> extends
|
|
15
|
-
/**
|
|
16
|
-
* The raw HTML content of the web page as a string.
|
|
17
|
-
*/
|
|
18
|
-
body: string;
|
|
19
|
-
/**
|
|
20
|
-
* The [Cheerio](https://cheerio.js.org/) object with parsed HTML.
|
|
21
|
-
* Cheerio is available only for HTML and XML content types.
|
|
22
|
-
*/
|
|
23
|
-
$: cheerio.CheerioAPI;
|
|
24
|
-
/**
|
|
25
|
-
* Wait for an element matching the selector to appear. Timeout is ignored.
|
|
26
|
-
*
|
|
27
|
-
* **Example usage:**
|
|
28
|
-
* ```ts
|
|
29
|
-
* async requestHandler({ waitForSelector, parseWithCheerio }) {
|
|
30
|
-
* await waitForSelector('article h1');
|
|
31
|
-
* const $ = await parseWithCheerio();
|
|
32
|
-
* const title = $('title').text();
|
|
33
|
-
* });
|
|
34
|
-
* ```
|
|
35
|
-
*/
|
|
36
|
-
waitForSelector(selector: string, timeoutMs?: number): Promise<void>;
|
|
37
|
-
/**
|
|
38
|
-
* Returns Cheerio handle, this is here to unify the crawler API, so they all have this handy method.
|
|
39
|
-
* It has the same return type as the `$` context property, use it only if you are abstracting your workflow to
|
|
40
|
-
* support different context types in one handler.
|
|
41
|
-
* When provided with the `selector` argument, it will throw if it's not available.
|
|
42
|
-
*
|
|
43
|
-
* **Example usage:**
|
|
44
|
-
* ```ts
|
|
45
|
-
* async requestHandler({ parseWithCheerio }) {
|
|
46
|
-
* const $ = await parseWithCheerio();
|
|
47
|
-
* const title = $('title').text();
|
|
48
|
-
* });
|
|
49
|
-
* ```
|
|
50
|
-
*/
|
|
51
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioRoot>;
|
|
52
|
-
/**
|
|
53
|
-
* Helper function for extracting URLs from the parsed HTML and adding them to the request queue.
|
|
54
|
-
*/
|
|
55
|
-
enqueueLinks(options?: EnqueueLinksOptions): Promise<BatchAddRequestsResult>;
|
|
15
|
+
JSONData extends Dictionary = any> extends DOMCrawlingContext<CheerioParseResult, UserData, JSONData> {
|
|
56
16
|
}
|
|
57
17
|
export type CheerioRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
58
18
|
JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserData, JSONData>>;
|
|
@@ -72,13 +32,15 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
|
|
|
72
32
|
* and then invokes the user-provided {@link CheerioCrawlerOptions.requestHandler} to extract page data
|
|
73
33
|
* using a [jQuery](https://jquery.com/)-like interface to the parsed HTML DOM.
|
|
74
34
|
*
|
|
75
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
76
|
-
* {@link
|
|
77
|
-
*
|
|
35
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
36
|
+
* {@link IRequestManager|request manager} provided via the {@link CheerioCrawlerOptions.requestManager|`requestManager`}
|
|
37
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
38
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
39
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
40
|
+
* result as `requestManager`.
|
|
78
41
|
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
42
|
+
* > The {@link CheerioCrawlerOptions.requestList|`requestList`} and {@link CheerioCrawlerOptions.requestQueue|`requestQueue`}
|
|
43
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
82
44
|
*
|
|
83
45
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
84
46
|
*
|
|
@@ -92,18 +54,18 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
|
|
|
92
54
|
* ]
|
|
93
55
|
* ```
|
|
94
56
|
*
|
|
95
|
-
* By default, `CheerioCrawler` only processes web pages with the `text/html`
|
|
96
|
-
* and `application/
|
|
57
|
+
* By default, `CheerioCrawler` only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
58
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
97
59
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
98
60
|
* use the {@link CheerioCrawlerOptions.additionalMimeTypes} constructor option.
|
|
99
61
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
100
62
|
* For more details, see {@link CheerioCrawlerOptions.requestHandler}.
|
|
101
63
|
*
|
|
102
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
103
|
-
*
|
|
104
|
-
*
|
|
105
|
-
*
|
|
106
|
-
* {@link
|
|
64
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
65
|
+
* {@link ConcurrencySystem}.
|
|
66
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
67
|
+
* `CheerioCrawler` constructor, or, for finer control, by injecting a pre-configured
|
|
68
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
107
69
|
*
|
|
108
70
|
* **Example usage:**
|
|
109
71
|
*
|
|
@@ -133,32 +95,12 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
|
|
|
133
95
|
* ```
|
|
134
96
|
* @category Crawlers
|
|
135
97
|
*/
|
|
136
|
-
export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension> extends
|
|
98
|
+
export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<CheerioCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends DOMCrawler<CheerioParseResult, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
137
99
|
/**
|
|
138
100
|
* All `CheerioCrawler` parameters are passed via an options object.
|
|
139
101
|
*/
|
|
140
|
-
constructor(options?: CheerioCrawlerOptions<ContextExtension, ExtendedContext
|
|
141
|
-
private parseContent;
|
|
142
|
-
private addHelpers;
|
|
143
|
-
}
|
|
144
|
-
interface EnqueueLinksInternalOptions {
|
|
145
|
-
options?: EnqueueLinksOptions;
|
|
146
|
-
$: cheerio.CheerioAPI | null;
|
|
147
|
-
requestQueue: RequestProvider;
|
|
148
|
-
robotsTxtFile?: RobotsTxtFile;
|
|
149
|
-
onSkippedRequest?: SkippedRequestCallback;
|
|
150
|
-
originalRequestUrl: string;
|
|
151
|
-
finalRequestUrl?: string;
|
|
152
|
-
}
|
|
153
|
-
interface BoundEnqueueLinksInternalOptions {
|
|
154
|
-
enqueueLinks: BasicCrawlingContext['enqueueLinks'];
|
|
155
|
-
options?: EnqueueLinksOptions;
|
|
156
|
-
$: cheerio.CheerioAPI | null;
|
|
157
|
-
originalRequestUrl: string;
|
|
158
|
-
finalRequestUrl?: string;
|
|
102
|
+
constructor(options?: CheerioCrawlerOptions<ContextExtension, ExtendedContext, any, any, Routes, StatisticStateExtension>);
|
|
159
103
|
}
|
|
160
|
-
/** @internal */
|
|
161
|
-
export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternalOptions | BoundEnqueueLinksInternalOptions): Promise<unknown>;
|
|
162
104
|
/**
|
|
163
105
|
* Creates new {@link Router} instance that works based on request labels.
|
|
164
106
|
* This instance can then serve as a `requestHandler` of your {@link CheerioCrawler}.
|
|
@@ -183,7 +125,6 @@ export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternal
|
|
|
183
125
|
* await crawler.run();
|
|
184
126
|
* ```
|
|
185
127
|
*/
|
|
186
|
-
|
|
187
|
-
export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData
|
|
188
|
-
export
|
|
189
|
-
//# sourceMappingURL=cheerio-crawler.d.ts.map
|
|
128
|
+
export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
129
|
+
export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
130
|
+
export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
@@ -1,7 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import * as cheerio from 'cheerio';
|
|
4
|
-
import { parseDocument } from 'htmlparser2';
|
|
1
|
+
import { DOMCrawler, Router } from '@crawlee/http';
|
|
2
|
+
import { cheerioParser } from './cheerio-parser.js';
|
|
5
3
|
/**
|
|
6
4
|
* Provides a framework for the parallel crawling of web pages using plain HTTP requests and
|
|
7
5
|
* [cheerio](https://www.npmjs.com/package/cheerio) HTML parser.
|
|
@@ -18,13 +16,15 @@ import { parseDocument } from 'htmlparser2';
|
|
|
18
16
|
* and then invokes the user-provided {@link CheerioCrawlerOptions.requestHandler} to extract page data
|
|
19
17
|
* using a [jQuery](https://jquery.com/)-like interface to the parsed HTML DOM.
|
|
20
18
|
*
|
|
21
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
22
|
-
* {@link
|
|
23
|
-
*
|
|
19
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
20
|
+
* {@link IRequestManager|request manager} provided via the {@link CheerioCrawlerOptions.requestManager|`requestManager`}
|
|
21
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
22
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
23
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
24
|
+
* result as `requestManager`.
|
|
24
25
|
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
26
|
+
* > The {@link CheerioCrawlerOptions.requestList|`requestList`} and {@link CheerioCrawlerOptions.requestQueue|`requestQueue`}
|
|
27
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
28
28
|
*
|
|
29
29
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
30
30
|
*
|
|
@@ -38,18 +38,18 @@ import { parseDocument } from 'htmlparser2';
|
|
|
38
38
|
* ]
|
|
39
39
|
* ```
|
|
40
40
|
*
|
|
41
|
-
* By default, `CheerioCrawler` only processes web pages with the `text/html`
|
|
42
|
-
* and `application/
|
|
41
|
+
* By default, `CheerioCrawler` only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
42
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
43
43
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
44
44
|
* use the {@link CheerioCrawlerOptions.additionalMimeTypes} constructor option.
|
|
45
45
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
46
46
|
* For more details, see {@link CheerioCrawlerOptions.requestHandler}.
|
|
47
47
|
*
|
|
48
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
* {@link
|
|
48
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
49
|
+
* {@link ConcurrencySystem}.
|
|
50
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
51
|
+
* `CheerioCrawler` constructor, or, for finer control, by injecting a pre-configured
|
|
52
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
53
53
|
*
|
|
54
54
|
* **Example usage:**
|
|
55
55
|
*
|
|
@@ -79,121 +79,14 @@ import { parseDocument } from 'htmlparser2';
|
|
|
79
79
|
* ```
|
|
80
80
|
* @category Crawlers
|
|
81
81
|
*/
|
|
82
|
-
export class CheerioCrawler extends
|
|
82
|
+
export class CheerioCrawler extends DOMCrawler {
|
|
83
83
|
/**
|
|
84
84
|
* All `CheerioCrawler` parameters are passed via an options object.
|
|
85
85
|
*/
|
|
86
|
-
constructor(options
|
|
87
|
-
super({
|
|
88
|
-
...options,
|
|
89
|
-
contextPipelineBuilder: () => this.buildContextPipeline()
|
|
90
|
-
.compose({
|
|
91
|
-
action: async (context) => await this.parseContent(context),
|
|
92
|
-
})
|
|
93
|
-
.compose({ action: async (context) => await this.addHelpers(context) }),
|
|
94
|
-
}, config);
|
|
86
|
+
constructor(options) {
|
|
87
|
+
super({ ...options, parser: cheerioParser() });
|
|
95
88
|
}
|
|
96
|
-
async parseContent(crawlingContext) {
|
|
97
|
-
const isXml = crawlingContext.contentType.type.includes('xml');
|
|
98
|
-
const body = Buffer.isBuffer(crawlingContext.body)
|
|
99
|
-
? crawlingContext.body.toString(crawlingContext.contentType.encoding)
|
|
100
|
-
: crawlingContext.body;
|
|
101
|
-
const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
|
|
102
|
-
const $ = cheerio.load(dom, {
|
|
103
|
-
xml: { decodeEntities: true, xmlMode: isXml },
|
|
104
|
-
});
|
|
105
|
-
return {
|
|
106
|
-
$,
|
|
107
|
-
body,
|
|
108
|
-
};
|
|
109
|
-
}
|
|
110
|
-
async addHelpers(crawlingContext) {
|
|
111
|
-
const originalEnqueueLinks = crawlingContext.enqueueLinks;
|
|
112
|
-
return {
|
|
113
|
-
enqueueLinks: async (enqueueOptions) => {
|
|
114
|
-
return (await cheerioCrawlerEnqueueLinks({
|
|
115
|
-
options: { ...enqueueOptions, limit: this.calculateEnqueuedRequestLimit(enqueueOptions?.limit) },
|
|
116
|
-
$: crawlingContext.$,
|
|
117
|
-
requestQueue: await this.getRequestQueue(),
|
|
118
|
-
robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
|
|
119
|
-
onSkippedRequest: this.handleSkippedRequest,
|
|
120
|
-
originalRequestUrl: crawlingContext.request.url,
|
|
121
|
-
finalRequestUrl: crawlingContext.request.loadedUrl,
|
|
122
|
-
enqueueLinks: originalEnqueueLinks,
|
|
123
|
-
})); // TODO make this type safe
|
|
124
|
-
},
|
|
125
|
-
waitForSelector: async (selector, _timeoutMs) => {
|
|
126
|
-
if (crawlingContext.$(selector).get().length === 0) {
|
|
127
|
-
throw new Error(`Selector '${selector}' not found.`);
|
|
128
|
-
}
|
|
129
|
-
},
|
|
130
|
-
parseWithCheerio: async (selector, timeoutMs) => {
|
|
131
|
-
if (selector) {
|
|
132
|
-
await crawlingContext.waitForSelector(selector, timeoutMs);
|
|
133
|
-
}
|
|
134
|
-
return crawlingContext.$;
|
|
135
|
-
},
|
|
136
|
-
};
|
|
137
|
-
}
|
|
138
|
-
}
|
|
139
|
-
/** @internal */
|
|
140
|
-
function containsEnqueueLinks(options) {
|
|
141
|
-
return !!options.enqueueLinks;
|
|
142
89
|
}
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
const { options: enqueueLinksOptions, $, originalRequestUrl, finalRequestUrl } = options;
|
|
146
|
-
if (!$) {
|
|
147
|
-
throw new Error('Cannot enqueue links because the DOM is not available.');
|
|
148
|
-
}
|
|
149
|
-
const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
|
|
150
|
-
enqueueStrategy: enqueueLinksOptions?.strategy,
|
|
151
|
-
finalRequestUrl,
|
|
152
|
-
originalRequestUrl,
|
|
153
|
-
userProvidedBaseUrl: enqueueLinksOptions?.baseUrl,
|
|
154
|
-
});
|
|
155
|
-
const urls = extractUrlsFromCheerio($, enqueueLinksOptions?.selector ?? 'a', enqueueLinksOptions?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
|
|
156
|
-
if (containsEnqueueLinks(options)) {
|
|
157
|
-
return options.enqueueLinks({
|
|
158
|
-
urls,
|
|
159
|
-
baseUrl,
|
|
160
|
-
...enqueueLinksOptions,
|
|
161
|
-
});
|
|
162
|
-
}
|
|
163
|
-
return enqueueLinks({
|
|
164
|
-
requestQueue: options.requestQueue,
|
|
165
|
-
robotsTxtFile: options.robotsTxtFile,
|
|
166
|
-
onSkippedRequest: options.onSkippedRequest,
|
|
167
|
-
urls,
|
|
168
|
-
baseUrl,
|
|
169
|
-
...enqueueLinksOptions,
|
|
170
|
-
});
|
|
171
|
-
}
|
|
172
|
-
/**
|
|
173
|
-
* Creates new {@link Router} instance that works based on request labels.
|
|
174
|
-
* This instance can then serve as a `requestHandler` of your {@link CheerioCrawler}.
|
|
175
|
-
* Defaults to the {@link CheerioCrawlingContext}.
|
|
176
|
-
*
|
|
177
|
-
* > Serves as a shortcut for using `Router.create<CheerioCrawlingContext>()`.
|
|
178
|
-
*
|
|
179
|
-
* ```ts
|
|
180
|
-
* import { CheerioCrawler, createCheerioRouter } from 'crawlee';
|
|
181
|
-
*
|
|
182
|
-
* const router = createCheerioRouter();
|
|
183
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
184
|
-
* ctx.log.info('...');
|
|
185
|
-
* });
|
|
186
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
187
|
-
* ctx.log.info('...');
|
|
188
|
-
* });
|
|
189
|
-
*
|
|
190
|
-
* const crawler = new CheerioCrawler({
|
|
191
|
-
* requestHandler: router,
|
|
192
|
-
* });
|
|
193
|
-
* await crawler.run();
|
|
194
|
-
* ```
|
|
195
|
-
*/
|
|
196
|
-
export function createCheerioRouter(routes) {
|
|
197
|
-
return Router.create(routes);
|
|
90
|
+
export function createCheerioRouter(routesOrSchemas) {
|
|
91
|
+
return Router.create(routesOrSchemas);
|
|
198
92
|
}
|
|
199
|
-
//# sourceMappingURL=cheerio-crawler.js.map
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { DOMParser } from '@crawlee/http';
|
|
2
|
+
import type { CheerioAPI } from 'cheerio';
|
|
3
|
+
export interface CheerioParseResult {
|
|
4
|
+
$: CheerioAPI;
|
|
5
|
+
body: string;
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* A {@link DOMParser} backed by [cheerio](https://www.npmjs.com/package/cheerio). Pass it to a
|
|
9
|
+
* {@link DOMCrawler} to get the crawling context {@link CheerioCrawler} provides.
|
|
10
|
+
*/
|
|
11
|
+
export declare function cheerioParser(): DOMParser<CheerioParseResult>;
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import { extractUrlsFromCheerio } from '@crawlee/utils/internal';
|
|
2
|
+
// slim uses faster htmlparser2 parser and doesn't load extra libs like undici
|
|
3
|
+
import * as cheerio from 'cheerio/slim';
|
|
4
|
+
import { parseDocument } from 'htmlparser2';
|
|
5
|
+
/**
|
|
6
|
+
* A {@link DOMParser} backed by [cheerio](https://www.npmjs.com/package/cheerio). Pass it to a
|
|
7
|
+
* {@link DOMCrawler} to get the crawling context {@link CheerioCrawler} provides.
|
|
8
|
+
*/
|
|
9
|
+
export function cheerioParser() {
|
|
10
|
+
return {
|
|
11
|
+
placeholderMembers: { $: true, body: true },
|
|
12
|
+
parse(context) {
|
|
13
|
+
const isXml = context.contentType.type.includes('xml');
|
|
14
|
+
const body = Buffer.isBuffer(context.body)
|
|
15
|
+
? context.body.toString(context.contentType.encoding)
|
|
16
|
+
: context.body;
|
|
17
|
+
const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
|
|
18
|
+
const $ = cheerio.load(dom, {
|
|
19
|
+
xml: { decodeEntities: true, xmlMode: isXml },
|
|
20
|
+
});
|
|
21
|
+
return { $, body };
|
|
22
|
+
},
|
|
23
|
+
extractLinks: ({ $ }, selector, baseUrl) => extractUrlsFromCheerio($, selector, baseUrl),
|
|
24
|
+
select: ({ $ }, selector) => $(selector).get(),
|
|
25
|
+
toCheerio: ({ $ }) => $,
|
|
26
|
+
};
|
|
27
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@crawlee/cheerio",
|
|
3
|
-
"version": "4.0.0-beta.
|
|
3
|
+
"version": "4.0.0-beta.190",
|
|
4
4
|
"description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
|
|
5
5
|
"engines": {
|
|
6
6
|
"node": ">=22.0.0"
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
},
|
|
39
39
|
"homepage": "https://crawlee.dev",
|
|
40
40
|
"scripts": {
|
|
41
|
-
"build": "
|
|
41
|
+
"build": "pnpm clean && pnpm compile && pnpm copy",
|
|
42
42
|
"clean": "rimraf ./dist",
|
|
43
43
|
"compile": "tsc -p tsconfig.build.json",
|
|
44
44
|
"copy": "tsx ../../scripts/copy.ts"
|
|
@@ -47,9 +47,9 @@
|
|
|
47
47
|
"access": "public"
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@crawlee/http": "4.0.0-beta.
|
|
51
|
-
"@crawlee/types": "4.0.0-beta.
|
|
52
|
-
"@crawlee/utils": "4.0.0-beta.
|
|
50
|
+
"@crawlee/http": "4.0.0-beta.190",
|
|
51
|
+
"@crawlee/types": "4.0.0-beta.190",
|
|
52
|
+
"@crawlee/utils": "4.0.0-beta.190",
|
|
53
53
|
"cheerio": "^1.0.0",
|
|
54
54
|
"htmlparser2": "^10.0.0",
|
|
55
55
|
"tslib": "^2.8.1"
|
|
@@ -61,5 +61,5 @@
|
|
|
61
61
|
}
|
|
62
62
|
}
|
|
63
63
|
},
|
|
64
|
-
"gitHead": "
|
|
64
|
+
"gitHead": "0a6af5ff4704225ddb009a7dd58bdeaad48ebb14"
|
|
65
65
|
}
|
package/index.d.ts.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,eAAe,CAAC;AAC9B,cAAc,gCAAgC,CAAC"}
|
package/index.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,eAAe,CAAC;AAC9B,cAAc,gCAAgC,CAAC"}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"cheerio-crawler.d.ts","sourceRoot":"","sources":["../../src/internals/cheerio-crawler.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EACR,oBAAoB,EACpB,aAAa,EACb,mBAAmB,EACnB,YAAY,EACZ,sBAAsB,EACtB,kBAAkB,EAClB,2BAA2B,EAC3B,gBAAgB,EAChB,cAAc,EACd,eAAe,EACf,YAAY,EACZ,sBAAsB,EACzB,MAAM,eAAe,CAAC;AACvB,OAAO,EAAgB,WAAW,EAAkD,MAAM,eAAe,CAAC;AAC1G,OAAO,KAAK,EAAE,sBAAsB,EAAE,UAAU,EAAE,MAAM,gBAAgB,CAAC;AACzE,OAAO,EAAE,KAAK,WAAW,EAA0B,KAAK,aAAa,EAAE,MAAM,gBAAgB,CAAC;AAE9F,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AAGnC,MAAM,MAAM,mBAAmB,CAC3B,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,YAAY,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAE7D,MAAM,WAAW,qBAAqB,CAClC,gBAAgB,GAAG,UAAU,CAAC,KAAK,CAAC,EACpC,eAAe,SAAS,sBAAsB,GAAG,sBAAsB,GAAG,gBAAgB,EAC1F,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,CACnC,SAAQ,kBAAkB,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,EAAE,gBAAgB,EAAE,eAAe,CAAC;CAAG;AAE9G,MAAM,MAAM,WAAW,CACnB,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,gBAAgB,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAEjE,MAAM,WAAW,sBAAsB,CACnC,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,CACnC,SAAQ,2BAA2B,CAAC,QAAQ,EAAE,QAAQ,CAAC;IACrD;;OAEG;IACH,IAAI,EAAE,MAAM,CAAC;IAEb;;;OAGG;IACH,CAAC,EAAE,OAAO,CAAC,UAAU,CAAC;IAEtB;;;;;;;;;;;OAWG;IACH,eAAe,CAAC,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAErE;;;;;;;;;;;;;OAaG;IACH,gBAAgB,CAAC,QAAQ,CAAC,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,WAAW,CAAC,CAAC;IAE9E;;OAEG;IACH,YAAY,CAAC,OAAO,CAAC,EAAE,mBAAmB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;CAChF;AAED,MAAM,MAAM,qBAAqB,CAC7B,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,cAAc,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAE/D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4EG;AACH,qBAAa,cAAc,CACvB,gBAAgB,GAAG,UAAU,CAAC,KAAK,CAAC,EACpC,eAAe,SAAS,sBAAsB,GAAG,sBAAsB,GAAG,gBAAgB,CAC5F,SAAQ,WAAW,CAAC,sBAAsB,EAAE,gBAAgB,EAAE,eAAe,CAAC;IAC5E;;OAEG;gBACS,OAAO,CAAC,EAAE,qBAAqB,CAAC,gBAAgB,EAAE,eAAe,CAAC,EAAE,MAAM,CAAC,EAAE,aAAa;YAexF,YAAY;YAgBZ,UAAU;CA8B3B;AAED,UAAU,2BAA2B;IACjC,OAAO,CAAC,EAAE,mBAAmB,CAAC;IAC9B,CAAC,EAAE,OAAO,CAAC,UAAU,GAAG,IAAI,CAAC;IAC7B,YAAY,EAAE,eAAe,CAAC;IAC9B,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,gBAAgB,CAAC,EAAE,sBAAsB,CAAC;IAC1C,kBAAkB,EAAE,MAAM,CAAC;IAC3B,eAAe,CAAC,EAAE,MAAM,CAAC;CAC5B;AAED,UAAU,gCAAgC;IACtC,YAAY,EAAE,oBAAoB,CAAC,cAAc,CAAC,CAAC;IACnD,OAAO,CAAC,EAAE,mBAAmB,CAAC;IAC9B,CAAC,EAAE,OAAO,CAAC,UAAU,GAAG,IAAI,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;IAC3B,eAAe,CAAC,EAAE,MAAM,CAAC;CAC5B;AASD,gBAAgB;AAChB,wBAAsB,0BAA0B,CAC5C,OAAO,EAAE,2BAA2B,GAAG,gCAAgC,oBAmC1E;AAED;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,mBAAmB,CAC/B,OAAO,SAAS,sBAAsB,GAAG,sBAAsB,EAC/D,QAAQ,SAAS,UAAU,GAAG,sBAAsB,CAAC,OAAO,CAAC,SAAS,CAAC,CAAC,EAC1E,MAAM,CAAC,EAAE,YAAY,CAAC,OAAO,EAAE,QAAQ,CAAC,kDAEzC"}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"cheerio-crawler.js","sourceRoot":"","sources":["../../src/internals/cheerio-crawler.ts"],"names":[],"mappings":"AAcA,OAAO,EAAE,YAAY,EAAE,WAAW,EAAE,sCAAsC,EAAE,MAAM,EAAE,MAAM,eAAe,CAAC;AAE1G,OAAO,EAAoB,sBAAsB,EAAsB,MAAM,gBAAgB,CAAC;AAE9F,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AACnC,OAAO,EAAE,aAAa,EAAE,MAAM,aAAa,CAAC;AA2E5C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4EG;AACH,MAAM,OAAO,cAGX,SAAQ,WAAsE;IAC5E;;OAEG;IACH,YAAY,OAAkE,EAAE,MAAsB;QAClG,KAAK,CACD;YACI,GAAG,OAAO;YACV,sBAAsB,EAAE,GAAG,EAAE,CACzB,IAAI,CAAC,oBAAoB,EAAE;iBACtB,OAAO,CAAC;gBACL,MAAM,EAAE,KAAK,EAAE,OAAO,EAAE,EAAE,CAAC,MAAM,IAAI,CAAC,YAAY,CAAC,OAAO,CAAC;aAC9D,CAAC;iBACD,OAAO,CAAC,EAAE,MAAM,EAAE,KAAK,EAAE,OAAO,EAAE,EAAE,CAAC,MAAM,IAAI,CAAC,UAAU,CAAC,OAAO,CAAC,EAAE,CAAC;SAClF,EACD,MAAM,CACT,CAAC;IACN,CAAC;IAEO,KAAK,CAAC,YAAY,CAAC,eAA4C;QACnE,MAAM,KAAK,GAAG,eAAe,CAAC,WAAW,CAAC,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC;QAC/D,MAAM,IAAI,GAAG,MAAM,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,CAAC;YAC9C,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC,QAAQ,CAAC,eAAe,CAAC,WAAW,CAAC,QAAQ,CAAC;YACrE,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC;QAC3B,MAAM,GAAG,GAAG,aAAa,CAAC,IAAI,EAAE,EAAE,cAAc,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,CAAC,CAAC;QAC1E,MAAM,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,GAAG,EAAE;YACxB,GAAG,EAAE,EAAE,cAAc,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE;SAC9B,CAAC,CAAC;QAErB,OAAO;YACH,CAAC;YACD,IAAI;SACP,CAAC;IACN,CAAC;IAEO,KAAK,CAAC,UAAU,CAAC,eAAgE;QACrF,MAAM,oBAAoB,GAAG,eAAe,CAAC,YAAY,CAAC;QAE1D,OAAO;YACH,YAAY,EAAE,KAAK,EAAE,cAAoC,EAAE,EAAE;gBACzD,OAAO,CAAC,MAAM,0BAA0B,CAAC;oBACrC,OAAO,EAAE,EAAE,GAAG,cAAc,EAAE,KAAK,EAAE,IAAI,CAAC,6BAA6B,CAAC,cAAc,EAAE,KAAK,CAAC,EAAE;oBAChG,CAAC,EAAE,eAAe,CAAC,CAAC;oBACpB,YAAY,EAAE,MAAM,IAAI,CAAC,eAAe,EAAE;oBAC1C,aAAa,EAAE,MAAM,IAAI,CAAC,sBAAsB,CAAC,eAAe,CAAC,OAAO,CAAC,GAAG,CAAC;oBAC7E,gBAAgB,EAAE,IAAI,CAAC,oBAAoB;oBAC3C,kBAAkB,EAAE,eAAe,CAAC,OAAO,CAAC,GAAG;oBAC/C,eAAe,EAAE,eAAe,CAAC,OAAO,CAAC,SAAS;oBAClD,YAAY,EAAE,oBAAoB;iBACrC,CAAC,CAA2B,CAAC,CAAC,2BAA2B;YAC9D,CAAC;YACD,eAAe,EAAE,KAAK,EAAE,QAAgB,EAAE,UAAmB,EAAE,EAAE;gBAC7D,IAAI,eAAe,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,GAAG,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;oBACjD,MAAM,IAAI,KAAK,CAAC,aAAa,QAAQ,cAAc,CAAC,CAAC;gBACzD,CAAC;YACL,CAAC;YACD,gBAAgB,EAAE,KAAK,EAAE,QAAiB,EAAE,SAAkB,EAAE,EAAE;gBAC9D,IAAI,QAAQ,EAAE,CAAC;oBACX,MAAM,eAAe,CAAC,eAAe,CAAC,QAAQ,EAAE,SAAS,CAAC,CAAC;gBAC/D,CAAC;gBAED,OAAO,eAAe,CAAC,CAAC,CAAC;YAC7B,CAAC;SACJ,CAAC;IACN,CAAC;CACJ;AAoBD,gBAAgB;AAChB,SAAS,oBAAoB,CACzB,OAAuE;IAEvE,OAAO,CAAC,CAAE,OAA4C,CAAC,YAAY,CAAC;AACxE,CAAC;AAED,gBAAgB;AAChB,MAAM,CAAC,KAAK,UAAU,0BAA0B,CAC5C,OAAuE;IAEvE,MAAM,EAAE,OAAO,EAAE,mBAAmB,EAAE,CAAC,EAAE,kBAAkB,EAAE,eAAe,EAAE,GAAG,OAAO,CAAC;IACzF,IAAI,CAAC,CAAC,EAAE,CAAC;QACL,MAAM,IAAI,KAAK,CAAC,wDAAwD,CAAC,CAAC;IAC9E,CAAC;IAED,MAAM,OAAO,GAAG,sCAAsC,CAAC;QACnD,eAAe,EAAE,mBAAmB,EAAE,QAAQ;QAC9C,eAAe;QACf,kBAAkB;QAClB,mBAAmB,EAAE,mBAAmB,EAAE,OAAO;KACpD,CAAC,CAAC;IAEH,MAAM,IAAI,GAAG,sBAAsB,CAC/B,CAAC,EACD,mBAAmB,EAAE,QAAQ,IAAI,GAAG,EACpC,mBAAmB,EAAE,OAAO,IAAI,eAAe,IAAI,kBAAkB,CACxE,CAAC;IAEF,IAAI,oBAAoB,CAAC,OAAO,CAAC,EAAE,CAAC;QAChC,OAAO,OAAO,CAAC,YAAY,CAAC;YACxB,IAAI;YACJ,OAAO;YACP,GAAG,mBAAmB;SACzB,CAAC,CAAC;IACP,CAAC;IACD,OAAO,YAAY,CAAC;QAChB,YAAY,EAAE,OAAO,CAAC,YAAY;QAClC,aAAa,EAAE,OAAO,CAAC,aAAa;QACpC,gBAAgB,EAAE,OAAO,CAAC,gBAAgB;QAC1C,IAAI;QACJ,OAAO;QACP,GAAG,mBAAmB;KACzB,CAAC,CAAC;AACP,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,MAAM,UAAU,mBAAmB,CAGjC,MAAwC;IACtC,OAAO,MAAM,CAAC,MAAM,CAAU,MAAM,CAAC,CAAC;AAC1C,CAAC"}
|