@crawlee/http 4.0.0-beta.184 → 4.0.0-beta.186
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/internals/dom-crawler.js +7 -12
- package/internals/file-download.js +11 -16
- package/internals/http-crawler.js +4 -8
- package/package.json +7 -7
package/internals/dom-crawler.js
CHANGED
|
@@ -23,22 +23,17 @@ export class DOMCrawler extends HttpCrawler {
|
|
|
23
23
|
buildContextPipeline() {
|
|
24
24
|
return super
|
|
25
25
|
.buildContextPipeline()
|
|
26
|
-
.compose(
|
|
27
|
-
|
|
28
|
-
cleanup: async (context) => {
|
|
29
|
-
// The `skipNavigation` placeholders below throw on access, so there is nothing to clean up.
|
|
30
|
-
if (!context.request.skipNavigation) {
|
|
31
|
-
await this.#parser.cleanup?.(context);
|
|
32
|
-
}
|
|
33
|
-
},
|
|
34
|
-
})
|
|
35
|
-
.compose({ action: async (context) => this.#addHelpers(context) });
|
|
26
|
+
.compose(this.#parseContent.bind(this))
|
|
27
|
+
.compose(async (context) => this.#addHelpers(context));
|
|
36
28
|
}
|
|
37
|
-
async #parseContent(context) {
|
|
29
|
+
async #parseContent(context, onCleanup) {
|
|
38
30
|
try {
|
|
39
|
-
|
|
31
|
+
const parsed = await this.#parser.parse(context);
|
|
32
|
+
onCleanup(() => this.#parser.cleanup?.(parsed));
|
|
33
|
+
return parsed;
|
|
40
34
|
}
|
|
41
35
|
catch (err) {
|
|
36
|
+
// The `skipNavigation` placeholders below throw on access, so there is nothing to clean up.
|
|
42
37
|
if (err instanceof NavigationSkippedError) {
|
|
43
38
|
return Object.defineProperties({}, Object.fromEntries(Object.keys(this.#parser.placeholderMembers).map((member) => [
|
|
44
39
|
member,
|
|
@@ -3,7 +3,6 @@ import { BasicCrawler } from '@crawlee/basic';
|
|
|
3
3
|
import { ResponseWithUrl } from '@crawlee/http-client';
|
|
4
4
|
import { Router } from '../index.js';
|
|
5
5
|
import { parseContentTypeFromResponse } from './utils.js';
|
|
6
|
-
const kBodyDrained = Symbol('bodyDrained');
|
|
7
6
|
/**
|
|
8
7
|
* Creates a transform stream that throws an error if the source data speed is below the specified minimum speed.
|
|
9
8
|
* This `Transform` checks the amount of data every `checkProgressInterval` milliseconds.
|
|
@@ -114,32 +113,28 @@ export class FileDownload extends BasicCrawler {
|
|
|
114
113
|
});
|
|
115
114
|
}
|
|
116
115
|
buildContextPipeline() {
|
|
117
|
-
return super.buildContextPipeline().compose(
|
|
118
|
-
action: async (context) => this.initiateDownload(context),
|
|
119
|
-
cleanup: async (context) => {
|
|
120
|
-
if (!context.response.bodyUsed) {
|
|
121
|
-
// Nobody consumed the body — cancel it so the
|
|
122
|
-
// underlying connection can be released.
|
|
123
|
-
await context.response.body?.cancel();
|
|
124
|
-
}
|
|
125
|
-
await context[kBodyDrained];
|
|
126
|
-
},
|
|
127
|
-
});
|
|
116
|
+
return super.buildContextPipeline().compose(this.initiateDownload.bind(this));
|
|
128
117
|
}
|
|
129
|
-
async initiateDownload(context) {
|
|
118
|
+
async initiateDownload(context, onCleanup) {
|
|
130
119
|
const response = await this.httpClient.sendRequest(context.request.intoFetchAPIRequest(), {
|
|
131
120
|
session: context.session,
|
|
132
121
|
});
|
|
133
122
|
const { type, charset: encoding } = parseContentTypeFromResponse(response);
|
|
134
123
|
context.request.url = response.url;
|
|
135
124
|
const { response: trackedResponse, bodyDrained } = trackBodyConsumption(response);
|
|
136
|
-
|
|
125
|
+
onCleanup(async () => {
|
|
126
|
+
if (!trackedResponse.bodyUsed) {
|
|
127
|
+
// Nobody consumed the body — cancel it so the
|
|
128
|
+
// underlying connection can be released.
|
|
129
|
+
await trackedResponse.body?.cancel();
|
|
130
|
+
}
|
|
131
|
+
await bodyDrained;
|
|
132
|
+
});
|
|
133
|
+
return {
|
|
137
134
|
request: context.request,
|
|
138
135
|
response: trackedResponse,
|
|
139
136
|
contentType: { type, encoding },
|
|
140
|
-
[kBodyDrained]: bodyDrained,
|
|
141
137
|
};
|
|
142
|
-
return contextExtension;
|
|
143
138
|
}
|
|
144
139
|
}
|
|
145
140
|
/**
|
|
@@ -182,9 +182,7 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
182
182
|
buildContextPipeline() {
|
|
183
183
|
// When navigation is skipped, `prepareHttpRequest` has already installed throwing getters for
|
|
184
184
|
// the response-derived members, so the guarded action is bypassed and the context left untouched.
|
|
185
|
-
const skipGuard = (action) => ({
|
|
186
|
-
action: async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {})),
|
|
187
|
-
});
|
|
185
|
+
const skipGuard = (action) => async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {}));
|
|
188
186
|
// A single navigation window covers the pre-navigation hooks, the navigation, and the post-navigation
|
|
189
187
|
// hooks: the whole phase shares one `navigationTimeoutSecs` budget, so a slow hook eats into the same
|
|
190
188
|
// window the navigation uses instead of each step being timed on its own.
|
|
@@ -196,9 +194,7 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
196
194
|
}
|
|
197
195
|
return addTimeoutToPromise(async () => step(ctx), remaining, navigationTimedOut);
|
|
198
196
|
});
|
|
199
|
-
let pipeline = ContextPipeline.create().compose(
|
|
200
|
-
action: this.prepareHttpRequest.bind(this),
|
|
201
|
-
});
|
|
197
|
+
let pipeline = ContextPipeline.create().compose(this.prepareHttpRequest.bind(this));
|
|
202
198
|
for (const hook of this.#preNavigationHooks) {
|
|
203
199
|
pipeline = pipeline.compose(windowGuard(hook));
|
|
204
200
|
}
|
|
@@ -207,8 +203,8 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
207
203
|
pipelineWithNavigation = pipelineWithNavigation.compose(windowGuard(hook));
|
|
208
204
|
}
|
|
209
205
|
return pipelineWithNavigation
|
|
210
|
-
.compose(
|
|
211
|
-
.compose(
|
|
206
|
+
.compose(this.processHttpResponse.bind(this))
|
|
207
|
+
.compose(this.handleBlockedRequestByContent.bind(this));
|
|
212
208
|
}
|
|
213
209
|
async prepareHttpRequest(crawlingContext) {
|
|
214
210
|
const { request } = crawlingContext;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@crawlee/http",
|
|
3
|
-
"version": "4.0.0-beta.
|
|
3
|
+
"version": "4.0.0-beta.186",
|
|
4
4
|
"description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
|
|
5
5
|
"engines": {
|
|
6
6
|
"node": ">=22.0.0"
|
|
@@ -49,11 +49,11 @@
|
|
|
49
49
|
"dependencies": {
|
|
50
50
|
"@apify/timeout": "^1.0.1",
|
|
51
51
|
"@apify/utilities": "^3.0.1",
|
|
52
|
-
"@crawlee/basic": "4.0.0-beta.
|
|
53
|
-
"@crawlee/core": "4.0.0-beta.
|
|
54
|
-
"@crawlee/http-client": "4.0.0-beta.
|
|
55
|
-
"@crawlee/types": "4.0.0-beta.
|
|
56
|
-
"@crawlee/utils": "4.0.0-beta.
|
|
52
|
+
"@crawlee/basic": "4.0.0-beta.186",
|
|
53
|
+
"@crawlee/core": "4.0.0-beta.186",
|
|
54
|
+
"@crawlee/http-client": "4.0.0-beta.186",
|
|
55
|
+
"@crawlee/types": "4.0.0-beta.186",
|
|
56
|
+
"@crawlee/utils": "4.0.0-beta.186",
|
|
57
57
|
"@types/content-type": "^1.1.8",
|
|
58
58
|
"cheerio": "^1.0.0",
|
|
59
59
|
"content-type": "^1.0.5",
|
|
@@ -70,5 +70,5 @@
|
|
|
70
70
|
}
|
|
71
71
|
}
|
|
72
72
|
},
|
|
73
|
-
"gitHead": "
|
|
73
|
+
"gitHead": "d15c77259254640d55d0492a196d1bc6a89ecd0b"
|
|
74
74
|
}
|