@crawlee/http 4.0.0-beta.20 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +1 -1
- package/index.js +1 -1
- package/internals/dom-crawler.d.ts +131 -0
- package/internals/dom-crawler.js +93 -0
- package/internals/file-download.d.ts +14 -50
- package/internals/file-download.js +42 -110
- package/internals/http-crawler.d.ts +216 -204
- package/internals/http-crawler.js +257 -233
- package/internals/utils.d.ts +11 -1
- package/internals/utils.js +50 -6
- package/package.json +13 -12
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/file-download.d.ts.map +0 -1
- package/internals/file-download.js.map +0 -1
- package/internals/http-crawler.d.ts.map +0 -1
- package/internals/http-crawler.js.map +0 -1
- package/internals/utils.d.ts.map +0 -1
- package/internals/utils.js.map +0 -1
|
@@ -1,27 +1,38 @@
|
|
|
1
1
|
import { Readable } from 'node:stream';
|
|
2
2
|
import util from 'node:util';
|
|
3
|
-
import { BasicCrawler,
|
|
4
|
-
import {
|
|
5
|
-
import
|
|
3
|
+
import { BasicCrawler, ContextPipeline, getCookiesFromResponse, NavigationSkippedError, remainingNavigationWindowMillis, RequestState, RequestThrottledError, Router, SessionError, } from '@crawlee/basic';
|
|
4
|
+
import { ResponseWithUrl } from '@crawlee/http-client';
|
|
5
|
+
import { parseArgument, RETRY_CSS_SELECTORS, schemas } from '@crawlee/utils/internal';
|
|
6
6
|
import contentTypeParser from 'content-type';
|
|
7
7
|
import iconv from 'iconv-lite';
|
|
8
|
-
import
|
|
9
|
-
import { addTimeoutToPromise, tryCancel } from '@apify/timeout';
|
|
10
|
-
import { parseContentTypeFromResponse } from './utils.js';
|
|
11
|
-
let TimeoutError;
|
|
8
|
+
import { z } from 'zod';
|
|
9
|
+
import { addTimeoutToPromise, storage, TimeoutError, tryCancel } from '@apify/timeout';
|
|
10
|
+
import { extractCharsetFromHtmlBytes, parseContentTypeFromResponse, processHttpRequestOptions } from './utils.js';
|
|
12
11
|
/**
|
|
13
12
|
* Default mime types, which HttpScraper supports.
|
|
14
13
|
*/
|
|
15
14
|
const HTML_AND_XML_MIME_TYPES = ['text/html', 'text/xml', 'application/xhtml+xml', 'application/xml'];
|
|
16
15
|
const APPLICATION_JSON_MIME_TYPE = 'application/json';
|
|
17
|
-
|
|
16
|
+
/**
|
|
17
|
+
* A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event
|
|
18
|
+
* loop. {@link HttpCrawler} folds these into the {@link ConcurrencySystem} it builds by default, with your own
|
|
19
|
+
* concurrency shortcuts (`minConcurrency`, `maxConcurrency`, `maxRequestsPerMinute`) kept on top.
|
|
20
|
+
*
|
|
21
|
+
* A {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default
|
|
22
|
+
* wholesale, tuning included, so spread these options in if you want to keep it:
|
|
23
|
+
*
|
|
24
|
+
* ```typescript
|
|
25
|
+
* new ConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, maxConcurrency: 50 });
|
|
26
|
+
* ```
|
|
27
|
+
*/
|
|
28
|
+
export const HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS = {
|
|
18
29
|
desiredConcurrency: 10,
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
30
|
+
loadSignals: {
|
|
31
|
+
eventLoop: {
|
|
32
|
+
snapshotIntervalSecs: 2,
|
|
33
|
+
maxBlockedMillis: 100,
|
|
34
|
+
overloadedRatio: 0.7,
|
|
35
|
+
},
|
|
25
36
|
},
|
|
26
37
|
};
|
|
27
38
|
/**
|
|
@@ -35,13 +46,15 @@ const HTTP_OPTIMIZED_AUTOSCALED_POOL_OPTIONS = {
|
|
|
35
46
|
*
|
|
36
47
|
* This crawler downloads each URL using a plain HTTP request and doesn't do any HTML parsing.
|
|
37
48
|
*
|
|
38
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
39
|
-
* {@link
|
|
40
|
-
*
|
|
49
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
50
|
+
* {@link IRequestManager|request manager} provided via the {@link HttpCrawlerOptions.requestManager|`requestManager`}
|
|
51
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
52
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
53
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
54
|
+
* result as `requestManager`.
|
|
41
55
|
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
56
|
+
* > The {@link HttpCrawlerOptions.requestList|`requestList`} and {@link HttpCrawlerOptions.requestQueue|`requestQueue`}
|
|
57
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
45
58
|
*
|
|
46
59
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
47
60
|
*
|
|
@@ -55,18 +68,18 @@ const HTTP_OPTIMIZED_AUTOSCALED_POOL_OPTIONS = {
|
|
|
55
68
|
* ]
|
|
56
69
|
* ```
|
|
57
70
|
*
|
|
58
|
-
* By default, this crawler only processes web pages with the `text/html`
|
|
59
|
-
* and `application/
|
|
71
|
+
* By default, this crawler only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
72
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
60
73
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
61
74
|
* use the {@link HttpCrawlerOptions.additionalMimeTypes} constructor option.
|
|
62
75
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
63
76
|
* For details, see {@link HttpCrawlerOptions.requestHandler}.
|
|
64
77
|
*
|
|
65
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
66
|
-
*
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
* {@link
|
|
78
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
79
|
+
* {@link ConcurrencySystem}.
|
|
80
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
81
|
+
* constructor, or, for finer control, by injecting a pre-configured
|
|
82
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
70
83
|
*
|
|
71
84
|
* **Example usage:**
|
|
72
85
|
*
|
|
@@ -92,108 +105,133 @@ const HTTP_OPTIMIZED_AUTOSCALED_POOL_OPTIONS = {
|
|
|
92
105
|
* @category Crawlers
|
|
93
106
|
*/
|
|
94
107
|
export class HttpCrawler extends BasicCrawler {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
108
|
+
// Internal storage uses the base (non-extended) context types. The public option types are
|
|
109
|
+
// extension-aware for consumer DX, but internally the pipeline composes hooks against the
|
|
110
|
+
// concrete crawling context, which does not statically carry `ContextExtension`. The members
|
|
111
|
+
// added by `extendContext` are present at runtime regardless.
|
|
112
|
+
#preNavigationHooks;
|
|
113
|
+
#postNavigationHooks;
|
|
114
|
+
#saveResponseCookies;
|
|
115
|
+
#navigationTimeoutMillis;
|
|
116
|
+
#ignoreTlsErrors;
|
|
117
|
+
#suggestResponseEncoding;
|
|
118
|
+
#forceResponseEncoding;
|
|
119
|
+
#supportedMimeTypes;
|
|
120
|
+
/**
|
|
121
|
+
* @internal
|
|
122
|
+
*/
|
|
106
123
|
static optionsShape = {
|
|
107
124
|
...BasicCrawler.optionsShape,
|
|
108
|
-
navigationTimeoutSecs:
|
|
109
|
-
|
|
110
|
-
additionalMimeTypes:
|
|
111
|
-
suggestResponseEncoding:
|
|
112
|
-
forceResponseEncoding:
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
preNavigationHooks: ow.optional.array,
|
|
117
|
-
postNavigationHooks: ow.optional.array,
|
|
125
|
+
navigationTimeoutSecs: schemas.anyNumber.default(30),
|
|
126
|
+
ignoreTlsErrors: z.boolean().default(true),
|
|
127
|
+
additionalMimeTypes: schemas.arrayOf(z.string(), 'strings').default(() => []),
|
|
128
|
+
suggestResponseEncoding: z.string().optional(),
|
|
129
|
+
forceResponseEncoding: z.string().optional(),
|
|
130
|
+
saveResponseCookies: z.boolean().default(true),
|
|
131
|
+
preNavigationHooks: schemas.anyArray.default(() => []),
|
|
132
|
+
postNavigationHooks: schemas.anyArray.default(() => []),
|
|
118
133
|
};
|
|
134
|
+
/** @internal */
|
|
135
|
+
static optionsSchema = z.strictObject(HttpCrawler.optionsShape);
|
|
119
136
|
/**
|
|
120
137
|
* All `HttpCrawlerOptions` parameters are passed via an options object.
|
|
121
138
|
*/
|
|
122
|
-
constructor(options = {}
|
|
123
|
-
|
|
124
|
-
const { navigationTimeoutSecs = 30, ignoreSslErrors = true, additionalMimeTypes = [], suggestResponseEncoding, forceResponseEncoding, persistCookiesPerSession, preNavigationHooks = [], postNavigationHooks = [], additionalHttpErrorStatusCodes = [], ignoreHttpErrorStatusCodes = [],
|
|
139
|
+
constructor(options = {}) {
|
|
140
|
+
const { navigationTimeoutSecs, ignoreTlsErrors, additionalMimeTypes, suggestResponseEncoding, forceResponseEncoding, saveResponseCookies, preNavigationHooks, postNavigationHooks,
|
|
125
141
|
// BasicCrawler
|
|
126
|
-
|
|
142
|
+
contextPipelineBuilder, ...basicCrawlerOptions } = parseArgument(options, HttpCrawler.optionsSchema, 'HttpCrawlerOptions');
|
|
127
143
|
super({
|
|
128
144
|
...basicCrawlerOptions,
|
|
129
|
-
autoscaledPoolOptions,
|
|
130
145
|
contextPipelineBuilder: contextPipelineBuilder ??
|
|
131
146
|
(() => this.buildContextPipeline()),
|
|
132
|
-
}
|
|
133
|
-
this
|
|
134
|
-
// Cookies should be persisted per session only if session pool is used
|
|
135
|
-
if (!this.useSessionPool && persistCookiesPerSession) {
|
|
136
|
-
throw new Error('You cannot use "persistCookiesPerSession" without "useSessionPool" set to true.');
|
|
137
|
-
}
|
|
138
|
-
this.supportedMimeTypes = new Set([...HTML_AND_XML_MIME_TYPES, APPLICATION_JSON_MIME_TYPE]);
|
|
147
|
+
});
|
|
148
|
+
this.#supportedMimeTypes = new Set([...HTML_AND_XML_MIME_TYPES, APPLICATION_JSON_MIME_TYPE]);
|
|
139
149
|
if (additionalMimeTypes.length)
|
|
140
|
-
this.
|
|
150
|
+
this.extendSupportedMimeTypes(additionalMimeTypes);
|
|
141
151
|
if (suggestResponseEncoding && forceResponseEncoding) {
|
|
142
152
|
this.log.warning('Both forceResponseEncoding and suggestResponseEncoding options are set. Using forceResponseEncoding.');
|
|
143
153
|
}
|
|
144
|
-
this
|
|
145
|
-
this
|
|
146
|
-
this
|
|
147
|
-
this
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
this
|
|
152
|
-
|
|
154
|
+
this.#navigationTimeoutMillis = navigationTimeoutSecs * 1000;
|
|
155
|
+
this.#ignoreTlsErrors = ignoreTlsErrors;
|
|
156
|
+
this.#suggestResponseEncoding = suggestResponseEncoding;
|
|
157
|
+
this.#forceResponseEncoding = forceResponseEncoding;
|
|
158
|
+
// Cast away the extension-aware option types to the base internal storage types (see the field
|
|
159
|
+
// declarations above). This is sound - the hooks only ever receive the base context plus the
|
|
160
|
+
// members `extendContext` added at runtime.
|
|
161
|
+
this.#preNavigationHooks = preNavigationHooks;
|
|
162
|
+
this.#postNavigationHooks = [
|
|
163
|
+
({ request, response }) => this.abortDownloadOfBody(request, response),
|
|
153
164
|
...postNavigationHooks,
|
|
154
165
|
];
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
166
|
+
this.#saveResponseCookies = saveResponseCookies;
|
|
167
|
+
}
|
|
168
|
+
/** @internal */
|
|
169
|
+
getNavigationTimeoutMillis() {
|
|
170
|
+
return this.#navigationTimeoutMillis;
|
|
171
|
+
}
|
|
172
|
+
/** @internal */
|
|
173
|
+
createDefaultConcurrencySystem(options) {
|
|
174
|
+
return super.createDefaultConcurrencySystem({
|
|
175
|
+
...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS,
|
|
176
|
+
...options,
|
|
177
|
+
});
|
|
161
178
|
}
|
|
162
179
|
buildContextPipeline() {
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
180
|
+
// When navigation is skipped, `prepareHttpRequest` has already installed throwing getters for
|
|
181
|
+
// the response-derived members, so the guarded action is bypassed and the context left untouched.
|
|
182
|
+
const skipGuard = (action) => async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {}));
|
|
183
|
+
// A single navigation window covers the pre-navigation hooks, the navigation, and the post-navigation
|
|
184
|
+
// hooks: the whole phase shares one `navigationTimeoutSecs` budget, so a slow hook eats into the same
|
|
185
|
+
// window the navigation uses instead of each step being timed on its own.
|
|
186
|
+
const navigationTimedOut = `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`;
|
|
187
|
+
const windowGuard = (step) => skipGuard(async (ctx) => {
|
|
188
|
+
const remaining = remainingNavigationWindowMillis(ctx, this.#navigationTimeoutMillis);
|
|
189
|
+
if (remaining <= 0) {
|
|
190
|
+
throw new TimeoutError(navigationTimedOut);
|
|
191
|
+
}
|
|
192
|
+
return addTimeoutToPromise(async () => step(ctx), remaining, navigationTimedOut);
|
|
193
|
+
});
|
|
194
|
+
let pipeline = ContextPipeline.create().compose(this.prepareHttpRequest.bind(this));
|
|
195
|
+
for (const hook of this.#preNavigationHooks) {
|
|
196
|
+
pipeline = pipeline.compose(windowGuard(hook));
|
|
197
|
+
}
|
|
198
|
+
let pipelineWithNavigation = pipeline.compose(skipGuard(this.makeHttpRequest.bind(this)));
|
|
199
|
+
for (const hook of this.#postNavigationHooks) {
|
|
200
|
+
pipelineWithNavigation = pipelineWithNavigation.compose(windowGuard(hook));
|
|
201
|
+
}
|
|
202
|
+
return pipelineWithNavigation
|
|
203
|
+
.compose(this.processHttpResponse.bind(this))
|
|
204
|
+
.compose(this.handleBlockedRequestByContent.bind(this));
|
|
169
205
|
}
|
|
170
|
-
async
|
|
171
|
-
const { request
|
|
206
|
+
async prepareHttpRequest(crawlingContext) {
|
|
207
|
+
const { request } = crawlingContext;
|
|
172
208
|
if (request.skipNavigation) {
|
|
173
209
|
return {
|
|
174
210
|
request: new Proxy(request, {
|
|
175
211
|
get(target, propertyName, receiver) {
|
|
176
212
|
if (propertyName === 'loadedUrl') {
|
|
177
|
-
throw new
|
|
213
|
+
throw new NavigationSkippedError('The `request.loadedUrl` property is not available - `skipNavigation` was used');
|
|
178
214
|
}
|
|
179
215
|
return Reflect.get(target, propertyName, receiver);
|
|
180
216
|
},
|
|
181
217
|
}),
|
|
182
218
|
get response() {
|
|
183
|
-
throw new
|
|
219
|
+
throw new NavigationSkippedError('The `response` property is not available - `skipNavigation` was used');
|
|
184
220
|
},
|
|
185
221
|
};
|
|
186
222
|
}
|
|
187
|
-
const preNavigationHooksCookies = this._getCookieHeaderFromRequest(request);
|
|
188
223
|
request.state = RequestState.BEFORE_NAV;
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
224
|
+
return {};
|
|
225
|
+
}
|
|
226
|
+
// oxlint-disable-next-line crawlee/prefer-private-fields -- patched by @crawlee/otel
|
|
227
|
+
async makeHttpRequest(crawlingContext) {
|
|
192
228
|
tryCancel();
|
|
193
|
-
const
|
|
194
|
-
const cookieString = this._applyCookies(crawlingContext, preNavigationHooksCookies, postNavigationHooksCookies);
|
|
229
|
+
const { request, session } = crawlingContext;
|
|
195
230
|
const proxyUrl = crawlingContext.proxyInfo?.url;
|
|
196
|
-
|
|
231
|
+
// Bound the request by whatever is left of the shared navigation window (the pre-navigation hooks may
|
|
232
|
+
// have already spent part of it), so it produces a clean navigation-timeout error rather than the raw
|
|
233
|
+
// client abort.
|
|
234
|
+
const httpResponse = await addTimeoutToPromise(async () => this.requestFunction({ request, session, proxyUrl }), Math.max(1, remainingNavigationWindowMillis(crawlingContext, this.#navigationTimeoutMillis)), `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
197
235
|
tryCancel();
|
|
198
236
|
request.loadedUrl = httpResponse?.url;
|
|
199
237
|
request.state = RequestState.AFTER_NAV;
|
|
@@ -203,46 +241,81 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
203
241
|
if (crawlingContext.request.skipNavigation) {
|
|
204
242
|
return {
|
|
205
243
|
get contentType() {
|
|
206
|
-
throw new
|
|
244
|
+
throw new NavigationSkippedError('The `contentType` property is not available - `skipNavigation` was used');
|
|
207
245
|
},
|
|
208
246
|
get body() {
|
|
209
|
-
throw new
|
|
247
|
+
throw new NavigationSkippedError('The `body` property is not available - `skipNavigation` was used');
|
|
210
248
|
},
|
|
211
249
|
get json() {
|
|
212
|
-
throw new
|
|
250
|
+
throw new NavigationSkippedError('The `json` property is not available - `skipNavigation` was used');
|
|
213
251
|
},
|
|
214
252
|
get waitForSelector() {
|
|
215
|
-
throw new
|
|
253
|
+
throw new NavigationSkippedError('The `waitForSelector` method is not available - `skipNavigation` was used');
|
|
216
254
|
},
|
|
217
255
|
get parseWithCheerio() {
|
|
218
|
-
throw new
|
|
256
|
+
throw new NavigationSkippedError('The `parseWithCheerio` method is not available - `skipNavigation` was used');
|
|
219
257
|
},
|
|
220
258
|
};
|
|
221
259
|
}
|
|
222
|
-
await this._executeHooks(this.postNavigationHooks, crawlingContext);
|
|
223
260
|
tryCancel();
|
|
224
|
-
|
|
261
|
+
// Before `parseResponse`, which throws for error status codes - a 429 the user opted into treating as an
|
|
262
|
+
// error is still a rate limit the domain should back off from.
|
|
263
|
+
if (crawlingContext.response.status === 429) {
|
|
264
|
+
const retryAfter = crawlingContext.response.headers.get('retry-after');
|
|
265
|
+
if (this.recordDomainRateLimit(crawlingContext.request.url, retryAfter)) {
|
|
266
|
+
// This is the one path that never reads the body, so cancel it to release the connection
|
|
267
|
+
// rather than leaving it to the garbage collector.
|
|
268
|
+
await crawlingContext.response.body?.cancel().catch(() => { });
|
|
269
|
+
throw new RequestThrottledError(`${crawlingContext.request.url} responded with 429.`);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
// Reading the body is still part of the navigation, so it draws from the same shared window: on a server
|
|
273
|
+
// that streams the body slowly the request completes (headers arrive) but the body read would otherwise
|
|
274
|
+
// run unbounded. `extendTimeout` from a post-navigation hook has already pushed this deadline out if asked.
|
|
275
|
+
const remaining = remainingNavigationWindowMillis(crawlingContext, this.#navigationTimeoutMillis);
|
|
276
|
+
if (remaining <= 0) {
|
|
277
|
+
throw new TimeoutError(`Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
278
|
+
}
|
|
279
|
+
const parsed = await addTimeoutToPromise(async () => this.parseResponse(crawlingContext.request, crawlingContext.response), remaining, `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
225
280
|
tryCancel();
|
|
226
281
|
const response = parsed.response;
|
|
227
282
|
const contentType = parsed.contentType;
|
|
283
|
+
const loadBody = async () => {
|
|
284
|
+
const { load } = await import('cheerio/slim');
|
|
285
|
+
return load(parsed.body.toString(), { xmlMode: contentType.type.includes('xml') });
|
|
286
|
+
};
|
|
228
287
|
const waitForSelector = async (selector, _timeoutMs) => {
|
|
229
|
-
const $ =
|
|
288
|
+
const $ = await loadBody();
|
|
230
289
|
if ($(selector).get().length === 0) {
|
|
231
290
|
throw new Error(`Selector '${selector}' not found.`);
|
|
232
291
|
}
|
|
233
292
|
};
|
|
234
293
|
const parseWithCheerio = async (selector, timeoutMs) => {
|
|
235
|
-
const $ =
|
|
294
|
+
const $ = await loadBody();
|
|
236
295
|
if (selector) {
|
|
237
296
|
await crawlingContext.waitForSelector(selector, timeoutMs);
|
|
238
297
|
}
|
|
239
298
|
return $;
|
|
240
299
|
};
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
300
|
+
this.throwOnBlockedRequest(response.status);
|
|
301
|
+
if (this.#saveResponseCookies) {
|
|
302
|
+
try {
|
|
303
|
+
for (const cookie of getCookiesFromResponse(response)) {
|
|
304
|
+
if (!cookie)
|
|
305
|
+
continue;
|
|
306
|
+
try {
|
|
307
|
+
await crawlingContext.session.cookieJar.setCookie(cookie, response.url, {
|
|
308
|
+
ignoreError: false,
|
|
309
|
+
});
|
|
310
|
+
}
|
|
311
|
+
catch (e) {
|
|
312
|
+
this.log.debug(`Could not set cookie: ${e.message}`);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
catch (e) {
|
|
317
|
+
this.log.exception(e, 'Could not get cookies from response');
|
|
318
|
+
}
|
|
246
319
|
}
|
|
247
320
|
return {
|
|
248
321
|
get json() {
|
|
@@ -259,13 +332,13 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
259
332
|
}
|
|
260
333
|
async handleBlockedRequestByContent(crawlingContext) {
|
|
261
334
|
if (this.retryOnBlocked) {
|
|
262
|
-
const error = await this
|
|
335
|
+
const error = await this.#isRequestBlocked(crawlingContext);
|
|
263
336
|
if (error)
|
|
264
337
|
throw new SessionError(error);
|
|
265
338
|
}
|
|
266
339
|
return {};
|
|
267
340
|
}
|
|
268
|
-
async isRequestBlocked(crawlingContext) {
|
|
341
|
+
async #isRequestBlocked(crawlingContext) {
|
|
269
342
|
if (HTML_AND_XML_MIME_TYPES.includes(crawlingContext.contentType.type)) {
|
|
270
343
|
const $ = await crawlingContext.parseWithCheerio();
|
|
271
344
|
const foundSelectors = RETRY_CSS_SELECTORS.filter((selector) => $(selector).length > 0);
|
|
@@ -273,47 +346,28 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
273
346
|
return `Found selectors: ${foundSelectors.join(', ')}`;
|
|
274
347
|
}
|
|
275
348
|
}
|
|
276
|
-
|
|
277
|
-
// eslint-disable-next-line dot-notation
|
|
278
|
-
(this.sessionPool?.['blockedStatusCodes'].length ?? 0) > 0
|
|
279
|
-
? // eslint-disable-next-line dot-notation
|
|
280
|
-
this.sessionPool['blockedStatusCodes']
|
|
281
|
-
: BLOCKED_STATUS_CODES;
|
|
282
|
-
if (blockedStatusCodes.includes(crawlingContext.response.status)) {
|
|
349
|
+
if (this.blockedStatusCodes.has(crawlingContext.response.status)) {
|
|
283
350
|
return `Blocked by status code ${crawlingContext.response.status}`;
|
|
284
351
|
}
|
|
285
352
|
return false;
|
|
286
353
|
}
|
|
287
|
-
/**
|
|
288
|
-
* Returns the `Cookie` header value based on the current context and
|
|
289
|
-
* any changes that occurred in the navigation hooks.
|
|
290
|
-
*/
|
|
291
|
-
_applyCookies({ session, request }, preHookCookies, postHookCookies) {
|
|
292
|
-
const sessionCookie = session?.getCookieString(request.url) ?? '';
|
|
293
|
-
const sourceCookies = [sessionCookie, preHookCookies, postHookCookies];
|
|
294
|
-
return mergeCookies(request.url, sourceCookies);
|
|
295
|
-
}
|
|
296
354
|
/**
|
|
297
355
|
* Function to make the HTTP request. It performs optimizations
|
|
298
356
|
* on the request such as only downloading the request body if the
|
|
299
357
|
* received content type matches text/html, application/xml, application/xhtml+xml.
|
|
300
358
|
*/
|
|
301
|
-
async
|
|
302
|
-
|
|
303
|
-
// @ts-ignore
|
|
304
|
-
({ TimeoutError } = await import('got-scraping'));
|
|
305
|
-
}
|
|
306
|
-
const opts = this._getRequestOptions(request, session, proxyUrl);
|
|
359
|
+
async requestFunction({ request, session, proxyUrl }) {
|
|
360
|
+
const opts = this.getRequestOptions(request, proxyUrl);
|
|
307
361
|
try {
|
|
308
|
-
return await this.
|
|
362
|
+
return await this.requestAsBrowser(opts, session);
|
|
309
363
|
}
|
|
310
364
|
catch (e) {
|
|
311
|
-
if (e instanceof TimeoutError) {
|
|
312
|
-
this.
|
|
313
|
-
return new Response(); // this will never happen, as
|
|
365
|
+
if (e instanceof Error && e.constructor.name === 'TimeoutError') {
|
|
366
|
+
this.handleRequestTimeout(session);
|
|
367
|
+
return new Response(); // this will never happen, as handleRequestTimeout always throws
|
|
314
368
|
}
|
|
315
369
|
if (this.isProxyError(e)) {
|
|
316
|
-
throw new SessionError(this.
|
|
370
|
+
throw new SessionError(this.getMessageFromError(e));
|
|
317
371
|
}
|
|
318
372
|
else {
|
|
319
373
|
throw e;
|
|
@@ -323,18 +377,15 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
323
377
|
/**
|
|
324
378
|
* Encodes and parses response according to the provided content type
|
|
325
379
|
*/
|
|
326
|
-
async
|
|
380
|
+
async parseResponse(request, response) {
|
|
327
381
|
const { status } = response;
|
|
328
382
|
const { type, charset } = parseContentTypeFromResponse(response);
|
|
329
|
-
const { response: reencodedResponse, encoding } = this._encodeResponse(request, response, charset);
|
|
330
|
-
const contentType = { type, encoding };
|
|
331
383
|
if (status >= 400 && status <= 599) {
|
|
332
|
-
this.
|
|
384
|
+
this.statistics.registerStatusCode(status);
|
|
333
385
|
}
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
const body = await reencodedResponse.text(); // TODO - this always uses UTF-8 (see https://developer.mozilla.org/en-US/docs/Web/API/Request/text)
|
|
386
|
+
if (this.isErrorStatusCode(status)) {
|
|
387
|
+
const { response: decoded } = this.encodeResponse(request, response, charset);
|
|
388
|
+
const body = await decoded.text(); // TODO - this always uses UTF-8 (see https://developer.mozilla.org/en-US/docs/Web/API/Request/text)
|
|
338
389
|
// Errors are often sent as JSON, so attempt to parse them,
|
|
339
390
|
// despite Accept header being set to text/html.
|
|
340
391
|
if (type === APPLICATION_JSON_MIME_TYPE) {
|
|
@@ -344,60 +395,60 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
344
395
|
message = util.inspect(errorResponse, { depth: 1, maxArrayLength: 10 });
|
|
345
396
|
throw new Error(`${status} - ${message}`);
|
|
346
397
|
}
|
|
347
|
-
if (
|
|
398
|
+
if (this.additionalHttpErrorStatusCodes.has(status)) {
|
|
348
399
|
throw new Error(`${status} - Error status code was set by user.`);
|
|
349
400
|
}
|
|
350
401
|
// It's not a JSON, so it's probably some text. Get the first 100 chars of it.
|
|
351
402
|
throw new Error(`${status} - Internal Server Error: ${body.slice(0, 100)}`);
|
|
352
403
|
}
|
|
353
|
-
|
|
354
|
-
|
|
404
|
+
if (HTML_AND_XML_MIME_TYPES.includes(type) && !charset && !this.#forceResponseEncoding) {
|
|
405
|
+
// The charset comes from the document itself, so the raw bytes are what we need -
|
|
406
|
+
// decoding them through `encodeResponse` first would consume the body for nothing.
|
|
407
|
+
const rawBytes = Buffer.from(await response.arrayBuffer());
|
|
408
|
+
const metaCharset = extractCharsetFromHtmlBytes(rawBytes);
|
|
409
|
+
const charsetToUse = metaCharset ?? this.#suggestResponseEncoding ?? 'utf-8';
|
|
410
|
+
const body = iconv.encodingExists(charsetToUse)
|
|
411
|
+
? iconv.decode(rawBytes, charsetToUse)
|
|
412
|
+
: rawBytes.toString('utf8');
|
|
413
|
+
return { response, contentType: { type, encoding: 'utf-8' }, body };
|
|
355
414
|
}
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
response,
|
|
361
|
-
contentType,
|
|
362
|
-
};
|
|
415
|
+
const { response: reencodedResponse, encoding } = this.encodeResponse(request, response, charset);
|
|
416
|
+
const contentType = { type, encoding };
|
|
417
|
+
if (HTML_AND_XML_MIME_TYPES.includes(type)) {
|
|
418
|
+
return { response, contentType, body: await reencodedResponse.text() };
|
|
363
419
|
}
|
|
420
|
+
return {
|
|
421
|
+
body: Buffer.from(await reencodedResponse.bytes()),
|
|
422
|
+
response,
|
|
423
|
+
contentType,
|
|
424
|
+
};
|
|
364
425
|
}
|
|
365
426
|
/**
|
|
366
427
|
* Combines the provided `requestOptions` with mandatory (non-overridable) values.
|
|
367
428
|
*/
|
|
368
|
-
|
|
429
|
+
getRequestOptions(request, proxyUrl) {
|
|
369
430
|
const requestOptions = {
|
|
370
431
|
url: request.url,
|
|
371
432
|
method: request.method,
|
|
372
433
|
proxyUrl,
|
|
373
|
-
timeout: this
|
|
374
|
-
cookieJar: this.persistCookiesPerSession ? session?.cookieJar : undefined,
|
|
375
|
-
sessionToken: session,
|
|
434
|
+
timeout: this.#navigationTimeoutMillis,
|
|
376
435
|
headers: request.headers,
|
|
377
|
-
https: {
|
|
378
|
-
rejectUnauthorized: !this.ignoreSslErrors,
|
|
379
|
-
},
|
|
380
436
|
body: undefined,
|
|
381
437
|
};
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
if (session?.proxyInfo?.ignoreTlsErrors) {
|
|
386
|
-
requestOptions.https = {
|
|
387
|
-
...requestOptions.https,
|
|
388
|
-
rejectUnauthorized: false,
|
|
389
|
-
};
|
|
438
|
+
if (requestOptions.headers?.cookie || requestOptions.headers?.Cookie) {
|
|
439
|
+
requestOptions.headers.Cookie = this.getCookieHeaderFromRequest(request);
|
|
440
|
+
delete requestOptions.headers.cookie;
|
|
390
441
|
}
|
|
391
442
|
if (/PATCH|POST|PUT/.test(request.method))
|
|
392
443
|
requestOptions.body = request.payload ?? '';
|
|
393
444
|
return requestOptions;
|
|
394
445
|
}
|
|
395
|
-
|
|
396
|
-
if (this
|
|
397
|
-
encoding = this
|
|
446
|
+
encodeResponse(request, response, encoding) {
|
|
447
|
+
if (this.#forceResponseEncoding) {
|
|
448
|
+
encoding = this.#forceResponseEncoding;
|
|
398
449
|
}
|
|
399
|
-
else if (!encoding && this
|
|
400
|
-
encoding = this
|
|
450
|
+
else if (!encoding && this.#suggestResponseEncoding) {
|
|
451
|
+
encoding = this.#suggestResponseEncoding;
|
|
401
452
|
}
|
|
402
453
|
// Fall back to utf-8 if we still don't have encoding.
|
|
403
454
|
const utf8 = 'utf8';
|
|
@@ -410,7 +461,9 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
410
461
|
// Try to re-encode a variety of unsupported encodings to utf-8
|
|
411
462
|
if (iconv.encodingExists(encoding)) {
|
|
412
463
|
const encodeStream = iconv.encodeStream(utf8);
|
|
413
|
-
const decodeStream = iconv
|
|
464
|
+
const decodeStream = iconv
|
|
465
|
+
.decodeStream(encoding)
|
|
466
|
+
.on('error', (err) => encodeStream.emit('error', err));
|
|
414
467
|
const reencodedBody = response.body
|
|
415
468
|
? Readable.toWeb(Readable.from(Readable.fromWeb(response.body)
|
|
416
469
|
.pipe(decodeStream)
|
|
@@ -426,15 +479,15 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
426
479
|
/**
|
|
427
480
|
* Checks and extends supported mime types
|
|
428
481
|
*/
|
|
429
|
-
|
|
482
|
+
extendSupportedMimeTypes(additionalMimeTypes) {
|
|
430
483
|
for (const mimeType of additionalMimeTypes) {
|
|
431
484
|
if (mimeType === '*/*') {
|
|
432
|
-
this
|
|
485
|
+
this.#supportedMimeTypes.add(mimeType);
|
|
433
486
|
continue;
|
|
434
487
|
}
|
|
435
488
|
try {
|
|
436
489
|
const parsedType = contentTypeParser.parse(mimeType);
|
|
437
|
-
this
|
|
490
|
+
this.#supportedMimeTypes.add(parsedType.type);
|
|
438
491
|
}
|
|
439
492
|
catch (err) {
|
|
440
493
|
throw new Error(`Can not parse mime type ${mimeType} from "options.additionalMimeTypes".`);
|
|
@@ -444,38 +497,39 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
444
497
|
/**
|
|
445
498
|
* Handles timeout request
|
|
446
499
|
*/
|
|
447
|
-
|
|
448
|
-
session
|
|
449
|
-
throw new Error(`
|
|
500
|
+
handleRequestTimeout(session) {
|
|
501
|
+
session.markBad();
|
|
502
|
+
throw new Error(`Request timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
450
503
|
}
|
|
451
|
-
|
|
504
|
+
abortDownloadOfBody(request, response) {
|
|
452
505
|
const { status } = response;
|
|
453
506
|
const { type } = parseContentTypeFromResponse(response);
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
// if we retry the request, can the Content-Type change?
|
|
457
|
-
const isTransientContentType = status >= 500 || blockedStatusCodes.includes(status);
|
|
458
|
-
if (!this.supportedMimeTypes.has(type) && !this.supportedMimeTypes.has('*/*') && !isTransientContentType) {
|
|
507
|
+
const isTransientContentType = status >= 500 || this.blockedStatusCodes.has(status);
|
|
508
|
+
if (!this.#supportedMimeTypes.has(type) && !this.#supportedMimeTypes.has('*/*') && !isTransientContentType) {
|
|
459
509
|
request.noRetry = true;
|
|
460
510
|
throw new Error(`Resource ${request.url} served Content-Type ${type}, ` +
|
|
461
|
-
`but only ${Array.from(this
|
|
511
|
+
`but only ${Array.from(this.#supportedMimeTypes).join(', ')} are allowed. Skipping resource.`);
|
|
462
512
|
}
|
|
463
513
|
}
|
|
464
514
|
/**
|
|
465
515
|
* @internal wraps public utility for mocking purposes
|
|
466
516
|
*/
|
|
467
|
-
|
|
517
|
+
requestAsBrowser = async (options, session) => {
|
|
468
518
|
const opts = processHttpRequestOptions({
|
|
469
519
|
...options,
|
|
470
|
-
cookieJar: options.cookieJar,
|
|
471
520
|
responseType: 'text',
|
|
472
521
|
});
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
522
|
+
// When saveResponseCookies is false, the response cookies must not mutate the
|
|
523
|
+
// session jar. Reads still go through the session (so session.setCookie() in pre-nav
|
|
524
|
+
// hooks keeps working) but a per-request clone is passed in so writes are discarded.
|
|
525
|
+
const cookieJar = this.#saveResponseCookies ? session.cookieJar : await session.cookieJar.clone();
|
|
526
|
+
// Bind the request to the shared navigation window instead of a fixed per-request timeout, so
|
|
527
|
+
// `extendTimeout()` can push the deadline and a fixed `AbortSignal.timeout` won't fire on its own and
|
|
528
|
+
// kill a lazily-read body mid-extension. This aborts the socket only during the header phase; the body
|
|
529
|
+
// read is bounded separately at the promise level (see `processHttpResponse`), so a slow-streaming body
|
|
530
|
+
// still fails cleanly with a navigation timeout, though the socket is left to close on its own.
|
|
531
|
+
const cancelSignal = storage.getStore()?.cancelTask.signal;
|
|
532
|
+
const response = await this.httpClient.sendRequest(new Request(opts.url, {
|
|
479
533
|
body: opts.body ? Readable.toWeb(opts.body) : undefined,
|
|
480
534
|
headers: new Headers(opts.headers),
|
|
481
535
|
method: opts.method,
|
|
@@ -483,45 +537,15 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
483
537
|
duplex: 'half',
|
|
484
538
|
}), {
|
|
485
539
|
session,
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
if (cookieStringRedirected !== '') {
|
|
492
|
-
updatedRequest.headers.set('Cookie', cookieStringRedirected);
|
|
493
|
-
}
|
|
494
|
-
}
|
|
495
|
-
},
|
|
540
|
+
cookieJar,
|
|
541
|
+
proxyUrl: opts.proxyUrl,
|
|
542
|
+
signal: cancelSignal,
|
|
543
|
+
timeoutMillis: cancelSignal ? undefined : opts.timeout,
|
|
544
|
+
ignoreTlsErrors: this.#ignoreTlsErrors,
|
|
496
545
|
});
|
|
497
546
|
return response;
|
|
498
547
|
};
|
|
499
548
|
}
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
* This instance can then serve as a `requestHandler` of your {@link HttpCrawler}.
|
|
503
|
-
* Defaults to the {@link HttpCrawlingContext}.
|
|
504
|
-
*
|
|
505
|
-
* > Serves as a shortcut for using `Router.create<HttpCrawlingContext>()`.
|
|
506
|
-
*
|
|
507
|
-
* ```ts
|
|
508
|
-
* import { HttpCrawler, createHttpRouter } from 'crawlee';
|
|
509
|
-
*
|
|
510
|
-
* const router = createHttpRouter();
|
|
511
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
512
|
-
* ctx.log.info('...');
|
|
513
|
-
* });
|
|
514
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
515
|
-
* ctx.log.info('...');
|
|
516
|
-
* });
|
|
517
|
-
*
|
|
518
|
-
* const crawler = new HttpCrawler({
|
|
519
|
-
* requestHandler: router,
|
|
520
|
-
* });
|
|
521
|
-
* await crawler.run();
|
|
522
|
-
* ```
|
|
523
|
-
*/
|
|
524
|
-
export function createHttpRouter(routes) {
|
|
525
|
-
return Router.create(routes);
|
|
549
|
+
export function createHttpRouter(routesOrSchemas) {
|
|
550
|
+
return Router.create(routesOrSchemas);
|
|
526
551
|
}
|
|
527
|
-
//# sourceMappingURL=http-crawler.js.map
|