@crawlee/http 4.0.0-beta.20 → 4.0.0-beta.201
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +1 -1
- package/index.js +1 -1
- package/internals/dom-crawler.d.ts +131 -0
- package/internals/dom-crawler.js +93 -0
- package/internals/file-download.d.ts +14 -50
- package/internals/file-download.js +42 -110
- package/internals/http-crawler.d.ts +216 -204
- package/internals/http-crawler.js +257 -233
- package/internals/utils.d.ts +11 -1
- package/internals/utils.js +50 -6
- package/package.json +13 -12
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/file-download.d.ts.map +0 -1
- package/internals/file-download.js.map +0 -1
- package/internals/http-crawler.d.ts.map +0 -1
- package/internals/http-crawler.js.map +0 -1
- package/internals/utils.d.ts.map +0 -1
- package/internals/utils.js.map +0 -1
package/README.md
CHANGED
|
@@ -1,23 +1,23 @@
|
|
|
1
1
|
<h1 align="center">
|
|
2
2
|
<a href="https://crawlee.dev">
|
|
3
3
|
<picture>
|
|
4
|
-
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true"
|
|
5
|
-
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500"
|
|
4
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true" />
|
|
5
|
+
<img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500" />
|
|
6
6
|
</picture>
|
|
7
7
|
</a>
|
|
8
|
-
<br
|
|
8
|
+
<br />
|
|
9
9
|
<small>A web scraping and browser automation library</small>
|
|
10
10
|
</h1>
|
|
11
11
|
|
|
12
|
-
<p align=center>
|
|
13
|
-
<a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift"
|
|
12
|
+
<p align="center">
|
|
13
|
+
<a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" width="250" height="55"/></a>
|
|
14
14
|
</p>
|
|
15
15
|
|
|
16
|
-
<p align=center>
|
|
17
|
-
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg"
|
|
18
|
-
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg"
|
|
19
|
-
<a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord"
|
|
20
|
-
<a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status"
|
|
16
|
+
<p align="center">
|
|
17
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" /></a>
|
|
18
|
+
<a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" /></a>
|
|
19
|
+
<a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" /></a>
|
|
20
|
+
<a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" /></a>
|
|
21
21
|
</p>
|
|
22
22
|
|
|
23
23
|
Crawlee covers your crawling and scraping end-to-end and **helps you build reliable scrapers. Fast.**
|
|
@@ -89,7 +89,7 @@ By default, Crawlee stores data to `./storage` in the current working directory.
|
|
|
89
89
|
We provide automated beta builds for every merged code change in Crawlee. You can find them in the npm [list of releases](https://www.npmjs.com/package/crawlee?activeTab=versions). If you want to test new features or bug fixes before we release them, feel free to install a beta build like this:
|
|
90
90
|
|
|
91
91
|
```bash
|
|
92
|
-
npm install crawlee@
|
|
92
|
+
npm install crawlee@next
|
|
93
93
|
```
|
|
94
94
|
|
|
95
95
|
If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need to specify dependency overrides in your `package.json` file so that you don't end up with multiple versions of Crawlee installed:
|
|
@@ -98,9 +98,9 @@ If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need
|
|
|
98
98
|
{
|
|
99
99
|
"overrides": {
|
|
100
100
|
"apify": {
|
|
101
|
-
"@crawlee/core": "
|
|
102
|
-
"@crawlee/types": "
|
|
103
|
-
"@crawlee/utils": "
|
|
101
|
+
"@crawlee/core": "$crawlee",
|
|
102
|
+
"@crawlee/types": "$crawlee",
|
|
103
|
+
"@crawlee/utils": "$crawlee"
|
|
104
104
|
}
|
|
105
105
|
}
|
|
106
106
|
}
|
package/index.d.ts
CHANGED
package/index.js
CHANGED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import type { AddRequestsBatchedResult, ContextPipeline, CrawlingContext, EnqueueLinksOptions, ExtractLinksOptions, GetUserDataFromRequest } from '@crawlee/basic';
|
|
2
|
+
import type { Awaitable, Dictionary } from '@crawlee/types';
|
|
3
|
+
import type { CheerioAPI } from 'cheerio';
|
|
4
|
+
import type { HttpCrawlerOptions, InternalHttpCrawlingContext } from './http-crawler.js';
|
|
5
|
+
import { HttpCrawler } from './http-crawler.js';
|
|
6
|
+
/**
|
|
7
|
+
* The minimum a {@link DOMParser} has to contribute to the crawling context - the serialized document, used by
|
|
8
|
+
* the {@link DOMCrawlingContext.parseWithCheerio|`parseWithCheerio`} helper.
|
|
9
|
+
*/
|
|
10
|
+
export interface DOMParseResult {
|
|
11
|
+
body: string;
|
|
12
|
+
}
|
|
13
|
+
/**
|
|
14
|
+
* Turns a response body into a DOM representation and knows how to query it. Passing one to {@link DOMCrawler}
|
|
15
|
+
* is what makes the crawler jsdom-based, linkedom-based, or based on a DOM implementation of your own.
|
|
16
|
+
*
|
|
17
|
+
* **Example usage:**
|
|
18
|
+
* ```ts
|
|
19
|
+
* import { DOMCrawler } from 'crawlee';
|
|
20
|
+
* import type { DOMParser } from 'crawlee';
|
|
21
|
+
*
|
|
22
|
+
* const myParser: DOMParser<{ body: string }> = {
|
|
23
|
+
* // ...
|
|
24
|
+
* };
|
|
25
|
+
*
|
|
26
|
+
* const crawler = new DOMCrawler({
|
|
27
|
+
* parser: myParser,
|
|
28
|
+
* async requestHandler({ body }) {
|
|
29
|
+
* // ...
|
|
30
|
+
* },
|
|
31
|
+
* });
|
|
32
|
+
* ```
|
|
33
|
+
*/
|
|
34
|
+
export interface DOMParser<Parsed extends DOMParseResult> {
|
|
35
|
+
/**
|
|
36
|
+
* The context members {@link DOMParser.parse|`parse`} contributes, mapped to `true`. Used to build the
|
|
37
|
+
* placeholders that report a helpful error when the members are accessed after `skipNavigation` - the `Record`
|
|
38
|
+
* type forces every key of `Parsed` to be listed, so the compiler catches an omission that would otherwise
|
|
39
|
+
* yield `undefined` (rather than throwing) after `skipNavigation`.
|
|
40
|
+
*/
|
|
41
|
+
readonly placeholderMembers: Record<keyof Parsed & string, true>;
|
|
42
|
+
parse(context: InternalHttpCrawlingContext): Awaitable<Parsed>;
|
|
43
|
+
/**
|
|
44
|
+
* Returns the URLs the `selector` matches, resolved against `baseUrl`.
|
|
45
|
+
*/
|
|
46
|
+
extractLinks(parsed: Parsed, selector: string, baseUrl: string): Awaitable<string[]>;
|
|
47
|
+
/**
|
|
48
|
+
* Returns the current matches of `selector`. Only the count is used, by
|
|
49
|
+
* {@link DOMCrawlingContext.waitForSelector|`waitForSelector`}.
|
|
50
|
+
*/
|
|
51
|
+
select(parsed: Parsed, selector: string): Awaitable<ArrayLike<unknown>>;
|
|
52
|
+
/**
|
|
53
|
+
* Whether the parse result can change after {@link DOMParser.parse|`parse`} returned - the case when the DOM
|
|
54
|
+
* implementation runs the page scripts. If it can, {@link DOMCrawlingContext.waitForSelector|`waitForSelector`}
|
|
55
|
+
* polls until the timeout elapses; otherwise it fails as soon as the selector does not match.
|
|
56
|
+
*/
|
|
57
|
+
readonly mutable?: boolean;
|
|
58
|
+
/**
|
|
59
|
+
* Returns a Cheerio handle over the parse result, for parsers that are backed by Cheerio anyway. Without it,
|
|
60
|
+
* {@link DOMCrawlingContext.parseWithCheerio|`parseWithCheerio`} parses
|
|
61
|
+
* {@link DOMParseResult.body|`body`} again.
|
|
62
|
+
*/
|
|
63
|
+
toCheerio?(parsed: Parsed): Awaitable<CheerioAPI>;
|
|
64
|
+
/**
|
|
65
|
+
* Releases whatever {@link DOMParser.parse|`parse`} allocated. Called after the request handler finishes or
|
|
66
|
+
* fails, and skipped entirely when navigation was skipped.
|
|
67
|
+
*/
|
|
68
|
+
cleanup?(parsed: Parsed): Awaitable<void>;
|
|
69
|
+
}
|
|
70
|
+
export interface DOMCrawlingHelpers {
|
|
71
|
+
/**
|
|
72
|
+
* Extracts URLs from the parsed DOM, without adding them to the request queue.
|
|
73
|
+
*/
|
|
74
|
+
extractLinks(options?: ExtractLinksOptions): Promise<string[]>;
|
|
75
|
+
/**
|
|
76
|
+
* Helper function for extracting URLs from the parsed DOM and adding them to the request queue.
|
|
77
|
+
*/
|
|
78
|
+
enqueueLinks(options?: EnqueueLinksOptions): Promise<AddRequestsBatchedResult>;
|
|
79
|
+
/**
|
|
80
|
+
* Wait for an element matching the selector to appear. The `timeoutMs` only has an effect when the parser is
|
|
81
|
+
* {@link DOMParser.mutable|`mutable`} (e.g. {@link JSDOMCrawler} with `runScripts: true`); otherwise the
|
|
82
|
+
* selector is checked once and the call resolves or throws immediately.
|
|
83
|
+
* Timeout defaults to 5s.
|
|
84
|
+
*
|
|
85
|
+
* **Example usage:**
|
|
86
|
+
* ```ts
|
|
87
|
+
* async requestHandler({ waitForSelector, parseWithCheerio }) {
|
|
88
|
+
* await waitForSelector('article h1');
|
|
89
|
+
* const $ = await parseWithCheerio();
|
|
90
|
+
* const title = $('title').text();
|
|
91
|
+
* });
|
|
92
|
+
* ```
|
|
93
|
+
*/
|
|
94
|
+
waitForSelector(selector: string, timeoutMs?: number): Promise<void>;
|
|
95
|
+
/**
|
|
96
|
+
* Returns Cheerio handle, allowing to work with the data same way as with {@link CheerioCrawler}.
|
|
97
|
+
* When provided with the `selector` argument, it will throw if it's not available.
|
|
98
|
+
*
|
|
99
|
+
* **Example usage:**
|
|
100
|
+
* ```javascript
|
|
101
|
+
* async requestHandler({ parseWithCheerio }) {
|
|
102
|
+
* const $ = await parseWithCheerio();
|
|
103
|
+
* const title = $('title').text();
|
|
104
|
+
* });
|
|
105
|
+
* ```
|
|
106
|
+
*/
|
|
107
|
+
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
108
|
+
}
|
|
109
|
+
export type DOMCrawlingContext<Parsed extends DOMParseResult = DOMParseResult, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
110
|
+
JSONData extends Dictionary = any> = InternalHttpCrawlingContext<UserData, JSONData> & Parsed & DOMCrawlingHelpers;
|
|
111
|
+
export interface DOMCrawlerOptions<Parsed extends DOMParseResult = DOMParseResult, ContextExtension = Dictionary<never>, ExtendedContext extends DOMCrawlingContext<Parsed> = DOMCrawlingContext<Parsed> & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, any>, StatisticStateExtension extends object = {}> extends Omit<HttpCrawlerOptions<DOMCrawlingContext<Parsed>, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>, 'contextPipelineBuilder'> {
|
|
112
|
+
/**
|
|
113
|
+
* The DOM implementation to parse the response bodies with. Its parse result becomes part of the crawling
|
|
114
|
+
* context, so the members the request handler receives follow from the parser you pass.
|
|
115
|
+
*/
|
|
116
|
+
parser: DOMParser<Parsed>;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* An {@link HttpCrawler} that parses each response into a DOM using the {@link DOMCrawlerOptions.parser|`parser`}
|
|
120
|
+
* it is given, and exposes the parse result plus the {@link DOMCrawlingContext.enqueueLinks|`enqueueLinks`} and
|
|
121
|
+
* {@link DOMCrawlingContext.extractLinks|`extractLinks`} helpers on the crawling context.
|
|
122
|
+
*
|
|
123
|
+
* {@link JSDOMCrawler} and {@link LinkeDOMCrawler} are this crawler with a parser already chosen.
|
|
124
|
+
*
|
|
125
|
+
* @category Crawlers
|
|
126
|
+
*/
|
|
127
|
+
export declare class DOMCrawler<Parsed extends DOMParseResult = DOMParseResult, ContextExtension = Dictionary<never>, ExtendedContext extends DOMCrawlingContext<Parsed> = DOMCrawlingContext<Parsed> & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<DOMCrawlingContext<Parsed>['request']>>, StatisticStateExtension extends object = {}> extends HttpCrawler<DOMCrawlingContext<Parsed>, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
128
|
+
#private;
|
|
129
|
+
constructor(options: DOMCrawlerOptions<Parsed, ContextExtension, ExtendedContext, Routes, StatisticStateExtension>);
|
|
130
|
+
protected buildContextPipeline(): ContextPipeline<CrawlingContext, DOMCrawlingContext<Parsed>>;
|
|
131
|
+
}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { EnqueueStrategy, NavigationSkippedError, resolveBaseUrlForEnqueueLinksFiltering } from '@crawlee/basic';
|
|
2
|
+
import { sleep } from '@crawlee/utils';
|
|
3
|
+
import { HttpCrawler } from './http-crawler.js';
|
|
4
|
+
/**
|
|
5
|
+
* An {@link HttpCrawler} that parses each response into a DOM using the {@link DOMCrawlerOptions.parser|`parser`}
|
|
6
|
+
* it is given, and exposes the parse result plus the {@link DOMCrawlingContext.enqueueLinks|`enqueueLinks`} and
|
|
7
|
+
* {@link DOMCrawlingContext.extractLinks|`extractLinks`} helpers on the crawling context.
|
|
8
|
+
*
|
|
9
|
+
* {@link JSDOMCrawler} and {@link LinkeDOMCrawler} are this crawler with a parser already chosen.
|
|
10
|
+
*
|
|
11
|
+
* @category Crawlers
|
|
12
|
+
*/
|
|
13
|
+
export class DOMCrawler extends HttpCrawler {
|
|
14
|
+
#parser;
|
|
15
|
+
constructor(options) {
|
|
16
|
+
const { parser, ...rest } = options;
|
|
17
|
+
super({ ...rest, contextPipelineBuilder: () => this.buildContextPipeline() });
|
|
18
|
+
this.#parser = parser;
|
|
19
|
+
}
|
|
20
|
+
buildContextPipeline() {
|
|
21
|
+
return super
|
|
22
|
+
.buildContextPipeline()
|
|
23
|
+
.compose(this.#parseContent.bind(this))
|
|
24
|
+
.compose(async (context) => this.#addHelpers(context));
|
|
25
|
+
}
|
|
26
|
+
async #parseContent(context, onCleanup) {
|
|
27
|
+
try {
|
|
28
|
+
const parsed = await this.#parser.parse(context);
|
|
29
|
+
onCleanup(() => this.#parser.cleanup?.(parsed));
|
|
30
|
+
return parsed;
|
|
31
|
+
}
|
|
32
|
+
catch (err) {
|
|
33
|
+
// The `skipNavigation` placeholders below throw on access, so there is nothing to clean up.
|
|
34
|
+
if (err instanceof NavigationSkippedError) {
|
|
35
|
+
return Object.defineProperties({}, Object.fromEntries(Object.keys(this.#parser.placeholderMembers).map((member) => [
|
|
36
|
+
member,
|
|
37
|
+
{
|
|
38
|
+
configurable: true,
|
|
39
|
+
enumerable: true,
|
|
40
|
+
get() {
|
|
41
|
+
throw new NavigationSkippedError(`The \`${member}\` property is not available - \`skipNavigation\` was used`, { cause: err });
|
|
42
|
+
},
|
|
43
|
+
},
|
|
44
|
+
])));
|
|
45
|
+
}
|
|
46
|
+
throw err;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
async #addHelpers(context) {
|
|
50
|
+
const { addRequests } = context;
|
|
51
|
+
const parser = this.#parser;
|
|
52
|
+
const extractLinks = async (options) => parser.extractLinks(context, options?.selector ?? 'a', options?.baseUrl ?? context.request.loadedUrl ?? context.request.url);
|
|
53
|
+
const waitForSelector = async (selector, timeoutMs = 5_000) => {
|
|
54
|
+
let remaining = parser.mutable ? timeoutMs : 0;
|
|
55
|
+
while ((await parser.select(context, selector)).length === 0) {
|
|
56
|
+
if (remaining <= 0) {
|
|
57
|
+
throw new Error(`Selector '${selector}' not found.`);
|
|
58
|
+
}
|
|
59
|
+
await sleep(50);
|
|
60
|
+
remaining -= 50;
|
|
61
|
+
}
|
|
62
|
+
};
|
|
63
|
+
return {
|
|
64
|
+
extractLinks,
|
|
65
|
+
waitForSelector,
|
|
66
|
+
enqueueLinks: async (options = {}) => {
|
|
67
|
+
const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
|
|
68
|
+
enqueueStrategy: options.strategy,
|
|
69
|
+
finalRequestUrl: context.request.loadedUrl,
|
|
70
|
+
originalRequestUrl: context.request.url,
|
|
71
|
+
userProvidedBaseUrl: options.baseUrl,
|
|
72
|
+
});
|
|
73
|
+
const urls = await extractLinks(options);
|
|
74
|
+
return addRequests(urls, {
|
|
75
|
+
...options,
|
|
76
|
+
baseUrl,
|
|
77
|
+
strategy: options.strategy ?? EnqueueStrategy.SameHostname,
|
|
78
|
+
});
|
|
79
|
+
},
|
|
80
|
+
async parseWithCheerio(selector, _timeoutMs = 5_000) {
|
|
81
|
+
// Import full cheerio (not cheerio/slim) to be browser-compliant for DOM Crawlers (not CheerioCrawler).
|
|
82
|
+
const $ = (await parser.toCheerio?.(context)) ??
|
|
83
|
+
(await import('cheerio')).load(context.body, {
|
|
84
|
+
xmlMode: context.contentType.type.includes('xml'),
|
|
85
|
+
});
|
|
86
|
+
if (selector && $(selector).get().length === 0) {
|
|
87
|
+
throw new Error(`Selector '${selector}' not found.`);
|
|
88
|
+
}
|
|
89
|
+
return $;
|
|
90
|
+
},
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
}
|
|
@@ -1,13 +1,11 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import type { BasicCrawlerOptions } from '@crawlee/basic';
|
|
1
|
+
import type { BasicCrawlerOptions, CrawlingContext, CrawlingRequest, LoadedRequest } from '@crawlee/basic';
|
|
3
2
|
import { BasicCrawler } from '@crawlee/basic';
|
|
4
|
-
import type { CrawlingContext, LoadedRequest, Request } from '@crawlee/core';
|
|
5
3
|
import type { Dictionary } from '@crawlee/types';
|
|
6
|
-
import type { ErrorHandler, GetUserDataFromRequest,
|
|
7
|
-
export type FileDownloadErrorHandler<UserData extends Dictionary = any
|
|
8
|
-
|
|
4
|
+
import type { ErrorHandler, GetUserDataFromRequest, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '../index.js';
|
|
5
|
+
export type FileDownloadErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
6
|
+
ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, FileDownloadCrawlingContext<UserData> & ContextExtension>;
|
|
9
7
|
export interface FileDownloadCrawlingContext<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
|
|
10
|
-
request: LoadedRequest<
|
|
8
|
+
request: LoadedRequest<CrawlingRequest<UserData>>;
|
|
11
9
|
response: Response;
|
|
12
10
|
contentType: {
|
|
13
11
|
type: string;
|
|
@@ -15,31 +13,6 @@ export interface FileDownloadCrawlingContext<UserData extends Dictionary = any>
|
|
|
15
13
|
};
|
|
16
14
|
}
|
|
17
15
|
export type FileDownloadRequestHandler<UserData extends Dictionary = any> = RequestHandler<FileDownloadCrawlingContext<UserData>>;
|
|
18
|
-
/**
|
|
19
|
-
* Creates a transform stream that throws an error if the source data speed is below the specified minimum speed.
|
|
20
|
-
* This `Transform` checks the amount of data every `checkProgressInterval` milliseconds.
|
|
21
|
-
* If the stream has received less than `minSpeedKbps * historyLengthMs / 1000` bytes in the last `historyLengthMs` milliseconds,
|
|
22
|
-
* it will throw an error.
|
|
23
|
-
*
|
|
24
|
-
* Can be used e.g. to abort a download if the network speed is too slow.
|
|
25
|
-
* @returns Transform stream that monitors the speed of the incoming data.
|
|
26
|
-
*/
|
|
27
|
-
export declare function MinimumSpeedStream({ minSpeedKbps, historyLengthMs, checkProgressInterval: checkProgressIntervalMs, }: {
|
|
28
|
-
minSpeedKbps: number;
|
|
29
|
-
historyLengthMs?: number;
|
|
30
|
-
checkProgressInterval?: number;
|
|
31
|
-
}): Transform;
|
|
32
|
-
/**
|
|
33
|
-
* Creates a transform stream that logs the progress of the incoming data.
|
|
34
|
-
* This `Transform` calls the `logProgress` function every `loggingInterval` milliseconds with the number of bytes received so far.
|
|
35
|
-
*
|
|
36
|
-
* Can be used e.g. to log the progress of a download.
|
|
37
|
-
* @returns Transform stream logging the progress of the incoming data.
|
|
38
|
-
*/
|
|
39
|
-
export declare function ByteCounterStream({ logTransferredBytes, loggingInterval, }: {
|
|
40
|
-
logTransferredBytes: (transferredBytes: number) => void;
|
|
41
|
-
loggingInterval?: number;
|
|
42
|
-
}): Transform;
|
|
43
16
|
/**
|
|
44
17
|
* Provides a framework for downloading files in parallel using plain HTTP requests. The URLs to download are fed either from a static list of URLs or they can be added on the fly from another crawler.
|
|
45
18
|
*
|
|
@@ -47,25 +20,15 @@ export declare function ByteCounterStream({ logTransferredBytes, loggingInterval
|
|
|
47
20
|
* However, it doesn't parse the content - if you need to e.g. extract data from the downloaded files,
|
|
48
21
|
* you might need to use {@link CheerioCrawler}, {@link PuppeteerCrawler} or {@link PlaywrightCrawler} instead.
|
|
49
22
|
*
|
|
50
|
-
* `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@link
|
|
23
|
+
* `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@link BasicCrawlerOptions.requestHandler} where the user can specify what to do with the downloaded data.
|
|
51
24
|
*
|
|
52
|
-
* The source URLs are represented using {@link Request} objects that are fed from {@link
|
|
25
|
+
* The source URLs are represented using {@link Request} objects that are fed from the {@link IRequestManager|request manager} provided via the {@link BasicCrawlerOptions.requestManager|`requestManager`} constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the result as `requestManager`.
|
|
53
26
|
*
|
|
54
|
-
*
|
|
27
|
+
* > The {@link BasicCrawlerOptions.requestList|`requestList`} and {@link BasicCrawlerOptions.requestQueue|`requestQueue`} options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
55
28
|
*
|
|
56
29
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
57
30
|
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
* ```
|
|
61
|
-
* preNavigationHooks: [
|
|
62
|
-
* (crawlingContext) => {
|
|
63
|
-
* // ...
|
|
64
|
-
* },
|
|
65
|
-
* ]
|
|
66
|
-
* ```
|
|
67
|
-
*
|
|
68
|
-
* New requests are only dispatched when there is enough free CPU and memory available, using the functionality provided by the {@link AutoscaledPool} class. All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions` parameter of the `FileCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency` {@link AutoscaledPool} options are available directly in the `FileCrawler` constructor.
|
|
31
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's {@link ConcurrencySystem}. Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the `FileCrawler` constructor, or, for finer control, by injecting a pre-configured {@link ConcurrencySystem|`concurrencySystem`}.
|
|
69
32
|
*
|
|
70
33
|
* ## Example usage
|
|
71
34
|
*
|
|
@@ -84,7 +47,8 @@ export declare function ByteCounterStream({ logTransferredBytes, loggingInterval
|
|
|
84
47
|
* ```
|
|
85
48
|
*/
|
|
86
49
|
export declare class FileDownload extends BasicCrawler<FileDownloadCrawlingContext> {
|
|
87
|
-
|
|
50
|
+
#private;
|
|
51
|
+
constructor(options?: Omit<BasicCrawlerOptions<FileDownloadCrawlingContext>, 'contextPipelineBuilder'>);
|
|
88
52
|
private initiateDownload;
|
|
89
53
|
}
|
|
90
54
|
/**
|
|
@@ -111,6 +75,6 @@ export declare class FileDownload extends BasicCrawler<FileDownloadCrawlingConte
|
|
|
111
75
|
* await crawler.run();
|
|
112
76
|
* ```
|
|
113
77
|
*/
|
|
114
|
-
|
|
115
|
-
export declare function createFileRouter<Context extends FileDownloadCrawlingContext = FileDownloadCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData
|
|
116
|
-
|
|
78
|
+
export declare function createFileRouter<Context extends FileDownloadCrawlingContext = FileDownloadCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
79
|
+
export declare function createFileRouter<Context extends FileDownloadCrawlingContext = FileDownloadCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
80
|
+
export declare function createFileRouter<Context extends FileDownloadCrawlingContext = FileDownloadCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
@@ -1,66 +1,7 @@
|
|
|
1
|
-
import { Transform } from 'node:stream';
|
|
2
|
-
import { finished } from 'node:stream/promises';
|
|
3
1
|
import { BasicCrawler, ContextPipeline } from '@crawlee/basic';
|
|
2
|
+
import { ResponseWithUrl } from '@crawlee/http-client';
|
|
4
3
|
import { Router } from '../index.js';
|
|
5
4
|
import { parseContentTypeFromResponse } from './utils.js';
|
|
6
|
-
/**
|
|
7
|
-
* Creates a transform stream that throws an error if the source data speed is below the specified minimum speed.
|
|
8
|
-
* This `Transform` checks the amount of data every `checkProgressInterval` milliseconds.
|
|
9
|
-
* If the stream has received less than `minSpeedKbps * historyLengthMs / 1000` bytes in the last `historyLengthMs` milliseconds,
|
|
10
|
-
* it will throw an error.
|
|
11
|
-
*
|
|
12
|
-
* Can be used e.g. to abort a download if the network speed is too slow.
|
|
13
|
-
* @returns Transform stream that monitors the speed of the incoming data.
|
|
14
|
-
*/
|
|
15
|
-
export function MinimumSpeedStream({ minSpeedKbps, historyLengthMs = 10e3, checkProgressInterval: checkProgressIntervalMs = 5e3, }) {
|
|
16
|
-
let snapshots = [];
|
|
17
|
-
const checkInterval = setInterval(() => {
|
|
18
|
-
const now = Date.now();
|
|
19
|
-
snapshots = snapshots.filter((snapshot) => now - snapshot.timestamp < historyLengthMs);
|
|
20
|
-
const totalBytes = snapshots.reduce((acc, snapshot) => acc + snapshot.bytes, 0);
|
|
21
|
-
const elapsed = (now - (snapshots[0]?.timestamp ?? 0)) / 1000;
|
|
22
|
-
if (totalBytes / 1024 / elapsed < minSpeedKbps) {
|
|
23
|
-
clearInterval(checkInterval);
|
|
24
|
-
stream.emit('error', new Error(`Stream speed too slow, aborting...`));
|
|
25
|
-
}
|
|
26
|
-
}, checkProgressIntervalMs);
|
|
27
|
-
const stream = new Transform({
|
|
28
|
-
transform: (chunk, _, callback) => {
|
|
29
|
-
snapshots.push({ timestamp: Date.now(), bytes: chunk.length });
|
|
30
|
-
callback(null, chunk);
|
|
31
|
-
},
|
|
32
|
-
final: (callback) => {
|
|
33
|
-
clearInterval(checkInterval);
|
|
34
|
-
callback();
|
|
35
|
-
},
|
|
36
|
-
});
|
|
37
|
-
return stream;
|
|
38
|
-
}
|
|
39
|
-
/**
|
|
40
|
-
* Creates a transform stream that logs the progress of the incoming data.
|
|
41
|
-
* This `Transform` calls the `logProgress` function every `loggingInterval` milliseconds with the number of bytes received so far.
|
|
42
|
-
*
|
|
43
|
-
* Can be used e.g. to log the progress of a download.
|
|
44
|
-
* @returns Transform stream logging the progress of the incoming data.
|
|
45
|
-
*/
|
|
46
|
-
export function ByteCounterStream({ logTransferredBytes, loggingInterval = 5000, }) {
|
|
47
|
-
let transferredBytes = 0;
|
|
48
|
-
let lastLogTime = Date.now();
|
|
49
|
-
return new Transform({
|
|
50
|
-
transform: (chunk, _, callback) => {
|
|
51
|
-
transferredBytes += chunk.length;
|
|
52
|
-
if (Date.now() - lastLogTime > loggingInterval) {
|
|
53
|
-
lastLogTime = Date.now();
|
|
54
|
-
logTransferredBytes(transferredBytes);
|
|
55
|
-
}
|
|
56
|
-
callback(null, chunk);
|
|
57
|
-
},
|
|
58
|
-
flush: (callback) => {
|
|
59
|
-
logTransferredBytes(transferredBytes);
|
|
60
|
-
callback();
|
|
61
|
-
},
|
|
62
|
-
});
|
|
63
|
-
}
|
|
64
5
|
/**
|
|
65
6
|
* Provides a framework for downloading files in parallel using plain HTTP requests. The URLs to download are fed either from a static list of URLs or they can be added on the fly from another crawler.
|
|
66
7
|
*
|
|
@@ -68,25 +9,15 @@ export function ByteCounterStream({ logTransferredBytes, loggingInterval = 5000,
|
|
|
68
9
|
* However, it doesn't parse the content - if you need to e.g. extract data from the downloaded files,
|
|
69
10
|
* you might need to use {@link CheerioCrawler}, {@link PuppeteerCrawler} or {@link PlaywrightCrawler} instead.
|
|
70
11
|
*
|
|
71
|
-
* `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@link
|
|
12
|
+
* `FileCrawler` downloads each URL using a plain HTTP request and then invokes the user-provided {@link BasicCrawlerOptions.requestHandler} where the user can specify what to do with the downloaded data.
|
|
72
13
|
*
|
|
73
|
-
* The source URLs are represented using {@link Request} objects that are fed from {@link
|
|
14
|
+
* The source URLs are represented using {@link Request} objects that are fed from the {@link IRequestManager|request manager} provided via the {@link BasicCrawlerOptions.requestManager|`requestManager`} constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the result as `requestManager`.
|
|
74
15
|
*
|
|
75
|
-
*
|
|
16
|
+
* > The {@link BasicCrawlerOptions.requestList|`requestList`} and {@link BasicCrawlerOptions.requestQueue|`requestQueue`} options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
76
17
|
*
|
|
77
18
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
78
19
|
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
* ```
|
|
82
|
-
* preNavigationHooks: [
|
|
83
|
-
* (crawlingContext) => {
|
|
84
|
-
* // ...
|
|
85
|
-
* },
|
|
86
|
-
* ]
|
|
87
|
-
* ```
|
|
88
|
-
*
|
|
89
|
-
* New requests are only dispatched when there is enough free CPU and memory available, using the functionality provided by the {@link AutoscaledPool} class. All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions` parameter of the `FileCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency` {@link AutoscaledPool} options are available directly in the `FileCrawler` constructor.
|
|
20
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's {@link ConcurrencySystem}. Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the `FileCrawler` constructor, or, for finer control, by injecting a pre-configured {@link ConcurrencySystem|`concurrencySystem`}.
|
|
90
21
|
*
|
|
91
22
|
* ## Example usage
|
|
92
23
|
*
|
|
@@ -109,53 +40,54 @@ export class FileDownload extends BasicCrawler {
|
|
|
109
40
|
constructor(options = {}) {
|
|
110
41
|
super({
|
|
111
42
|
...options,
|
|
112
|
-
contextPipelineBuilder: () =>
|
|
113
|
-
action: async (context) => this.initiateDownload(context),
|
|
114
|
-
cleanup: async (context) => {
|
|
115
|
-
await (context.response.body ? finished(context.response.body) : Promise.resolve());
|
|
116
|
-
},
|
|
117
|
-
}),
|
|
43
|
+
contextPipelineBuilder: () => this.#buildContextPipeline(),
|
|
118
44
|
});
|
|
119
45
|
}
|
|
120
|
-
|
|
121
|
-
|
|
46
|
+
#buildContextPipeline() {
|
|
47
|
+
return ContextPipeline.create().compose(this.initiateDownload.bind(this));
|
|
48
|
+
}
|
|
49
|
+
async initiateDownload(context, onCleanup) {
|
|
50
|
+
const response = await this.httpClient.sendRequest(context.request.intoFetchAPIRequest(), {
|
|
122
51
|
session: context.session,
|
|
123
52
|
});
|
|
124
53
|
const { type, charset: encoding } = parseContentTypeFromResponse(response);
|
|
125
54
|
context.request.url = response.url;
|
|
126
|
-
const
|
|
55
|
+
const { response: trackedResponse, bodyDrained } = trackBodyConsumption(response);
|
|
56
|
+
onCleanup(async () => {
|
|
57
|
+
if (!trackedResponse.bodyUsed) {
|
|
58
|
+
// Nobody consumed the body — cancel it so the
|
|
59
|
+
// underlying connection can be released.
|
|
60
|
+
await trackedResponse.body?.cancel();
|
|
61
|
+
}
|
|
62
|
+
await bodyDrained;
|
|
63
|
+
});
|
|
64
|
+
return {
|
|
127
65
|
request: context.request,
|
|
128
|
-
response,
|
|
66
|
+
response: trackedResponse,
|
|
129
67
|
contentType: { type, encoding },
|
|
130
68
|
};
|
|
131
|
-
return contextExtension;
|
|
132
69
|
}
|
|
133
70
|
}
|
|
134
71
|
/**
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
* > Serves as a shortcut for using `Router.create<FileDownloadCrawlingContext>()`.
|
|
140
|
-
*
|
|
141
|
-
* ```ts
|
|
142
|
-
* import { FileDownload, createFileRouter } from 'crawlee';
|
|
143
|
-
*
|
|
144
|
-
* const router = createFileRouter();
|
|
145
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
146
|
-
* ctx.log.info('...');
|
|
147
|
-
* });
|
|
148
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
149
|
-
* ctx.log.info('...');
|
|
150
|
-
* });
|
|
151
|
-
*
|
|
152
|
-
* const crawler = new FileDownload({
|
|
153
|
-
* requestHandler: router,
|
|
154
|
-
* });
|
|
155
|
-
* await crawler.run();
|
|
156
|
-
* ```
|
|
72
|
+
* Wraps a Response so that we can track when the body stream has been fully
|
|
73
|
+
* consumed (or errored). Pipes the original body through a TransformStream;
|
|
74
|
+
* the readable side becomes the new Response body, and `pipeTo` gives us a
|
|
75
|
+
* promise that resolves once the body is fully read or cancelled.
|
|
157
76
|
*/
|
|
158
|
-
|
|
159
|
-
|
|
77
|
+
function trackBodyConsumption(response) {
|
|
78
|
+
if (!response.body) {
|
|
79
|
+
return { response, bodyDrained: Promise.resolve() };
|
|
80
|
+
}
|
|
81
|
+
const passthrough = new TransformStream();
|
|
82
|
+
const bodyDrained = response.body.pipeTo(passthrough.writable).catch(() => { });
|
|
83
|
+
const trackedResponse = new ResponseWithUrl(passthrough.readable, {
|
|
84
|
+
headers: response.headers,
|
|
85
|
+
status: response.status,
|
|
86
|
+
statusText: response.statusText,
|
|
87
|
+
url: response.url,
|
|
88
|
+
});
|
|
89
|
+
return { response: trackedResponse, bodyDrained };
|
|
90
|
+
}
|
|
91
|
+
export function createFileRouter(routesOrSchemas) {
|
|
92
|
+
return Router.create(routesOrSchemas);
|
|
160
93
|
}
|
|
161
|
-
//# sourceMappingURL=file-download.js.map
|