@crawlee/http 4.0.0-beta.2 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +1 -1
- package/index.js +1 -1
- package/internals/dom-crawler.d.ts +131 -0
- package/internals/dom-crawler.js +93 -0
- package/internals/file-download.d.ts +23 -46
- package/internals/file-download.js +53 -109
- package/internals/http-crawler.d.ts +226 -293
- package/internals/http-crawler.js +344 -429
- package/internals/utils.d.ts +19 -0
- package/internals/utils.js +79 -0
- package/package.json +13 -12
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/file-download.d.ts.map +0 -1
- package/internals/file-download.js.map +0 -1
- package/internals/http-crawler.d.ts.map +0 -1
- package/internals/http-crawler.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
|
@@ -1,28 +1,38 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { Readable } from 'node:stream';
|
|
2
2
|
import util from 'node:util';
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import
|
|
3
|
+
import { BasicCrawler, ContextPipeline, getCookiesFromResponse, NavigationSkippedError, remainingNavigationWindowMillis, RequestState, RequestThrottledError, Router, SessionError, } from '@crawlee/basic';
|
|
4
|
+
import { ResponseWithUrl } from '@crawlee/http-client';
|
|
5
|
+
import { parseArgument, RETRY_CSS_SELECTORS, schemas } from '@crawlee/utils/internal';
|
|
6
6
|
import contentTypeParser from 'content-type';
|
|
7
7
|
import iconv from 'iconv-lite';
|
|
8
|
-
import
|
|
9
|
-
import
|
|
10
|
-
import {
|
|
11
|
-
import { concatStreamToBuffer, readStreamToString } from '@apify/utilities';
|
|
12
|
-
let TimeoutError;
|
|
8
|
+
import { z } from 'zod';
|
|
9
|
+
import { addTimeoutToPromise, storage, TimeoutError, tryCancel } from '@apify/timeout';
|
|
10
|
+
import { extractCharsetFromHtmlBytes, parseContentTypeFromResponse, processHttpRequestOptions } from './utils.js';
|
|
13
11
|
/**
|
|
14
12
|
* Default mime types, which HttpScraper supports.
|
|
15
13
|
*/
|
|
16
14
|
const HTML_AND_XML_MIME_TYPES = ['text/html', 'text/xml', 'application/xhtml+xml', 'application/xml'];
|
|
17
15
|
const APPLICATION_JSON_MIME_TYPE = 'application/json';
|
|
18
|
-
|
|
16
|
+
/**
|
|
17
|
+
* A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event
|
|
18
|
+
* loop. {@link HttpCrawler} folds these into the {@link ConcurrencySystem} it builds by default, with your own
|
|
19
|
+
* concurrency shortcuts (`minConcurrency`, `maxConcurrency`, `maxRequestsPerMinute`) kept on top.
|
|
20
|
+
*
|
|
21
|
+
* A {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default
|
|
22
|
+
* wholesale, tuning included, so spread these options in if you want to keep it:
|
|
23
|
+
*
|
|
24
|
+
* ```typescript
|
|
25
|
+
* new ConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, maxConcurrency: 50 });
|
|
26
|
+
* ```
|
|
27
|
+
*/
|
|
28
|
+
export const HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS = {
|
|
19
29
|
desiredConcurrency: 10,
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
30
|
+
loadSignals: {
|
|
31
|
+
eventLoop: {
|
|
32
|
+
snapshotIntervalSecs: 2,
|
|
33
|
+
maxBlockedMillis: 100,
|
|
34
|
+
overloadedRatio: 0.7,
|
|
35
|
+
},
|
|
26
36
|
},
|
|
27
37
|
};
|
|
28
38
|
/**
|
|
@@ -36,38 +46,40 @@ const HTTP_OPTIMIZED_AUTOSCALED_POOL_OPTIONS = {
|
|
|
36
46
|
*
|
|
37
47
|
* This crawler downloads each URL using a plain HTTP request and doesn't do any HTML parsing.
|
|
38
48
|
*
|
|
39
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
40
|
-
* {@link
|
|
41
|
-
*
|
|
49
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
50
|
+
* {@link IRequestManager|request manager} provided via the {@link HttpCrawlerOptions.requestManager|`requestManager`}
|
|
51
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
52
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
53
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
54
|
+
* result as `requestManager`.
|
|
42
55
|
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
56
|
+
* > The {@link HttpCrawlerOptions.requestList|`requestList`} and {@link HttpCrawlerOptions.requestQueue|`requestQueue`}
|
|
57
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
46
58
|
*
|
|
47
59
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
48
60
|
*
|
|
49
|
-
* We can use the `preNavigationHooks` to adjust
|
|
61
|
+
* We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
|
|
50
62
|
*
|
|
51
63
|
* ```javascript
|
|
52
64
|
* preNavigationHooks: [
|
|
53
|
-
* (crawlingContext
|
|
65
|
+
* (crawlingContext) => {
|
|
54
66
|
* // ...
|
|
55
67
|
* },
|
|
56
68
|
* ]
|
|
57
69
|
* ```
|
|
58
70
|
*
|
|
59
|
-
* By default, this crawler only processes web pages with the `text/html`
|
|
60
|
-
* and `application/
|
|
71
|
+
* By default, this crawler only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
72
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
61
73
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
62
74
|
* use the {@link HttpCrawlerOptions.additionalMimeTypes} constructor option.
|
|
63
75
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
64
76
|
* For details, see {@link HttpCrawlerOptions.requestHandler}.
|
|
65
77
|
*
|
|
66
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
* {@link
|
|
78
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
79
|
+
* {@link ConcurrencySystem}.
|
|
80
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
81
|
+
* constructor, or, for finer control, by injecting a pre-configured
|
|
82
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
71
83
|
*
|
|
72
84
|
* **Example usage:**
|
|
73
85
|
*
|
|
@@ -93,184 +105,240 @@ const HTTP_OPTIMIZED_AUTOSCALED_POOL_OPTIONS = {
|
|
|
93
105
|
* @category Crawlers
|
|
94
106
|
*/
|
|
95
107
|
export class HttpCrawler extends BasicCrawler {
|
|
96
|
-
|
|
108
|
+
// Internal storage uses the base (non-extended) context types. The public option types are
|
|
109
|
+
// extension-aware for consumer DX, but internally the pipeline composes hooks against the
|
|
110
|
+
// concrete crawling context, which does not statically carry `ContextExtension`. The members
|
|
111
|
+
// added by `extendContext` are present at runtime regardless.
|
|
112
|
+
#preNavigationHooks;
|
|
113
|
+
#postNavigationHooks;
|
|
114
|
+
#saveResponseCookies;
|
|
115
|
+
#navigationTimeoutMillis;
|
|
116
|
+
#ignoreTlsErrors;
|
|
117
|
+
#suggestResponseEncoding;
|
|
118
|
+
#forceResponseEncoding;
|
|
119
|
+
#supportedMimeTypes;
|
|
97
120
|
/**
|
|
98
|
-
*
|
|
99
|
-
* Only available if used by the crawler.
|
|
121
|
+
* @internal
|
|
100
122
|
*/
|
|
101
|
-
proxyConfiguration;
|
|
102
|
-
userRequestHandlerTimeoutMillis;
|
|
103
|
-
preNavigationHooks;
|
|
104
|
-
postNavigationHooks;
|
|
105
|
-
persistCookiesPerSession;
|
|
106
|
-
navigationTimeoutMillis;
|
|
107
|
-
ignoreSslErrors;
|
|
108
|
-
suggestResponseEncoding;
|
|
109
|
-
forceResponseEncoding;
|
|
110
|
-
additionalHttpErrorStatusCodes;
|
|
111
|
-
ignoreHttpErrorStatusCodes;
|
|
112
|
-
supportedMimeTypes;
|
|
113
123
|
static optionsShape = {
|
|
114
124
|
...BasicCrawler.optionsShape,
|
|
115
|
-
navigationTimeoutSecs:
|
|
116
|
-
|
|
117
|
-
additionalMimeTypes:
|
|
118
|
-
suggestResponseEncoding:
|
|
119
|
-
forceResponseEncoding:
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
ignoreHttpErrorStatusCodes: ow.optional.array.ofType(ow.number),
|
|
124
|
-
preNavigationHooks: ow.optional.array,
|
|
125
|
-
postNavigationHooks: ow.optional.array,
|
|
125
|
+
navigationTimeoutSecs: schemas.anyNumber.default(30),
|
|
126
|
+
ignoreTlsErrors: z.boolean().default(true),
|
|
127
|
+
additionalMimeTypes: schemas.arrayOf(z.string(), 'strings').default(() => []),
|
|
128
|
+
suggestResponseEncoding: z.string().optional(),
|
|
129
|
+
forceResponseEncoding: z.string().optional(),
|
|
130
|
+
saveResponseCookies: z.boolean().default(true),
|
|
131
|
+
preNavigationHooks: schemas.anyArray.default(() => []),
|
|
132
|
+
postNavigationHooks: schemas.anyArray.default(() => []),
|
|
126
133
|
};
|
|
134
|
+
/** @internal */
|
|
135
|
+
static optionsSchema = z.strictObject(HttpCrawler.optionsShape);
|
|
127
136
|
/**
|
|
128
137
|
* All `HttpCrawlerOptions` parameters are passed via an options object.
|
|
129
138
|
*/
|
|
130
|
-
constructor(options = {}
|
|
131
|
-
|
|
132
|
-
const { requestHandler, requestHandlerTimeoutSecs = 60, navigationTimeoutSecs = 30, ignoreSslErrors = true, additionalMimeTypes = [], suggestResponseEncoding, forceResponseEncoding, proxyConfiguration, persistCookiesPerSession, preNavigationHooks = [], postNavigationHooks = [], additionalHttpErrorStatusCodes = [], ignoreHttpErrorStatusCodes = [],
|
|
139
|
+
constructor(options = {}) {
|
|
140
|
+
const { navigationTimeoutSecs, ignoreTlsErrors, additionalMimeTypes, suggestResponseEncoding, forceResponseEncoding, saveResponseCookies, preNavigationHooks, postNavigationHooks,
|
|
133
141
|
// BasicCrawler
|
|
134
|
-
|
|
142
|
+
contextPipelineBuilder, ...basicCrawlerOptions } = parseArgument(options, HttpCrawler.optionsSchema, 'HttpCrawlerOptions');
|
|
135
143
|
super({
|
|
136
144
|
...basicCrawlerOptions,
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
requestHandlerTimeoutSecs: navigationTimeoutSecs + requestHandlerTimeoutSecs + BASIC_CRAWLER_TIMEOUT_BUFFER_SECS,
|
|
142
|
-
}, config);
|
|
143
|
-
this.config = config;
|
|
144
|
-
// FIXME any
|
|
145
|
-
this.requestHandler = requestHandler ?? this.router;
|
|
146
|
-
// Cookies should be persisted per session only if session pool is used
|
|
147
|
-
if (!this.useSessionPool && persistCookiesPerSession) {
|
|
148
|
-
throw new Error('You cannot use "persistCookiesPerSession" without "useSessionPool" set to true.');
|
|
149
|
-
}
|
|
150
|
-
this.supportedMimeTypes = new Set([...HTML_AND_XML_MIME_TYPES, APPLICATION_JSON_MIME_TYPE]);
|
|
145
|
+
contextPipelineBuilder: contextPipelineBuilder ??
|
|
146
|
+
(() => this.buildContextPipeline()),
|
|
147
|
+
});
|
|
148
|
+
this.#supportedMimeTypes = new Set([...HTML_AND_XML_MIME_TYPES, APPLICATION_JSON_MIME_TYPE]);
|
|
151
149
|
if (additionalMimeTypes.length)
|
|
152
|
-
this.
|
|
150
|
+
this.extendSupportedMimeTypes(additionalMimeTypes);
|
|
153
151
|
if (suggestResponseEncoding && forceResponseEncoding) {
|
|
154
152
|
this.log.warning('Both forceResponseEncoding and suggestResponseEncoding options are set. Using forceResponseEncoding.');
|
|
155
153
|
}
|
|
156
|
-
this
|
|
157
|
-
this
|
|
158
|
-
this
|
|
159
|
-
this
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
this
|
|
164
|
-
this
|
|
165
|
-
|
|
166
|
-
({ request, response }) => this._abortDownloadOfBody(request, response),
|
|
154
|
+
this.#navigationTimeoutMillis = navigationTimeoutSecs * 1000;
|
|
155
|
+
this.#ignoreTlsErrors = ignoreTlsErrors;
|
|
156
|
+
this.#suggestResponseEncoding = suggestResponseEncoding;
|
|
157
|
+
this.#forceResponseEncoding = forceResponseEncoding;
|
|
158
|
+
// Cast away the extension-aware option types to the base internal storage types (see the field
|
|
159
|
+
// declarations above). This is sound - the hooks only ever receive the base context plus the
|
|
160
|
+
// members `extendContext` added at runtime.
|
|
161
|
+
this.#preNavigationHooks = preNavigationHooks;
|
|
162
|
+
this.#postNavigationHooks = [
|
|
163
|
+
({ request, response }) => this.abortDownloadOfBody(request, response),
|
|
167
164
|
...postNavigationHooks,
|
|
168
165
|
];
|
|
169
|
-
|
|
170
|
-
|
|
166
|
+
this.#saveResponseCookies = saveResponseCookies;
|
|
167
|
+
}
|
|
168
|
+
/** @internal */
|
|
169
|
+
getNavigationTimeoutMillis() {
|
|
170
|
+
return this.#navigationTimeoutMillis;
|
|
171
|
+
}
|
|
172
|
+
/** @internal */
|
|
173
|
+
createDefaultConcurrencySystem(options) {
|
|
174
|
+
return super.createDefaultConcurrencySystem({
|
|
175
|
+
...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS,
|
|
176
|
+
...options,
|
|
177
|
+
});
|
|
178
|
+
}
|
|
179
|
+
buildContextPipeline() {
|
|
180
|
+
// When navigation is skipped, `prepareHttpRequest` has already installed throwing getters for
|
|
181
|
+
// the response-derived members, so the guarded action is bypassed and the context left untouched.
|
|
182
|
+
const skipGuard = (action) => async (ctx) => (ctx.request.skipNavigation ? {} : ((await action(ctx)) ?? {}));
|
|
183
|
+
// A single navigation window covers the pre-navigation hooks, the navigation, and the post-navigation
|
|
184
|
+
// hooks: the whole phase shares one `navigationTimeoutSecs` budget, so a slow hook eats into the same
|
|
185
|
+
// window the navigation uses instead of each step being timed on its own.
|
|
186
|
+
const navigationTimedOut = `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`;
|
|
187
|
+
const windowGuard = (step) => skipGuard(async (ctx) => {
|
|
188
|
+
const remaining = remainingNavigationWindowMillis(ctx, this.#navigationTimeoutMillis);
|
|
189
|
+
if (remaining <= 0) {
|
|
190
|
+
throw new TimeoutError(navigationTimedOut);
|
|
191
|
+
}
|
|
192
|
+
return addTimeoutToPromise(async () => step(ctx), remaining, navigationTimedOut);
|
|
193
|
+
});
|
|
194
|
+
let pipeline = ContextPipeline.create().compose(this.prepareHttpRequest.bind(this));
|
|
195
|
+
for (const hook of this.#preNavigationHooks) {
|
|
196
|
+
pipeline = pipeline.compose(windowGuard(hook));
|
|
171
197
|
}
|
|
172
|
-
|
|
173
|
-
|
|
198
|
+
let pipelineWithNavigation = pipeline.compose(skipGuard(this.makeHttpRequest.bind(this)));
|
|
199
|
+
for (const hook of this.#postNavigationHooks) {
|
|
200
|
+
pipelineWithNavigation = pipelineWithNavigation.compose(windowGuard(hook));
|
|
174
201
|
}
|
|
202
|
+
return pipelineWithNavigation
|
|
203
|
+
.compose(this.processHttpResponse.bind(this))
|
|
204
|
+
.compose(this.handleBlockedRequestByContent.bind(this));
|
|
175
205
|
}
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
// Test if the property can be configured on the crawler
|
|
193
|
-
throw new Error(`${extension.name} tries to set property "${key}" that is not configurable on ${className} instance.`);
|
|
194
|
-
}
|
|
195
|
-
if (!isSameType && exists) {
|
|
196
|
-
// Assuming that extensions will only add up configuration
|
|
197
|
-
throw new Error(`${extension.name} tries to set property of different type "${extensionType}". "${className}.${key}: ${originalType}".`);
|
|
198
|
-
}
|
|
199
|
-
this.log.warning(`${extension.name} is overriding "${className}.${key}: ${originalType}" with ${value}.`);
|
|
200
|
-
this[key] = value;
|
|
206
|
+
async prepareHttpRequest(crawlingContext) {
|
|
207
|
+
const { request } = crawlingContext;
|
|
208
|
+
if (request.skipNavigation) {
|
|
209
|
+
return {
|
|
210
|
+
request: new Proxy(request, {
|
|
211
|
+
get(target, propertyName, receiver) {
|
|
212
|
+
if (propertyName === 'loadedUrl') {
|
|
213
|
+
throw new NavigationSkippedError('The `request.loadedUrl` property is not available - `skipNavigation` was used');
|
|
214
|
+
}
|
|
215
|
+
return Reflect.get(target, propertyName, receiver);
|
|
216
|
+
},
|
|
217
|
+
}),
|
|
218
|
+
get response() {
|
|
219
|
+
throw new NavigationSkippedError('The `response` property is not available - `skipNavigation` was used');
|
|
220
|
+
},
|
|
221
|
+
};
|
|
201
222
|
}
|
|
223
|
+
request.state = RequestState.BEFORE_NAV;
|
|
224
|
+
return {};
|
|
202
225
|
}
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
async _runRequestHandler(crawlingContext) {
|
|
226
|
+
// oxlint-disable-next-line crawlee/prefer-private-fields -- patched by @crawlee/otel
|
|
227
|
+
async makeHttpRequest(crawlingContext) {
|
|
228
|
+
tryCancel();
|
|
207
229
|
const { request, session } = crawlingContext;
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
230
|
+
const proxyUrl = crawlingContext.proxyInfo?.url;
|
|
231
|
+
// Bound the request by whatever is left of the shared navigation window (the pre-navigation hooks may
|
|
232
|
+
// have already spent part of it), so it produces a clean navigation-timeout error rather than the raw
|
|
233
|
+
// client abort.
|
|
234
|
+
const httpResponse = await addTimeoutToPromise(async () => this.requestFunction({ request, session, proxyUrl }), Math.max(1, remainingNavigationWindowMillis(crawlingContext, this.#navigationTimeoutMillis)), `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
235
|
+
tryCancel();
|
|
236
|
+
request.loadedUrl = httpResponse?.url;
|
|
237
|
+
request.state = RequestState.AFTER_NAV;
|
|
238
|
+
return { request: request, response: httpResponse };
|
|
239
|
+
}
|
|
240
|
+
async processHttpResponse(crawlingContext) {
|
|
241
|
+
if (crawlingContext.request.skipNavigation) {
|
|
242
|
+
return {
|
|
243
|
+
get contentType() {
|
|
244
|
+
throw new NavigationSkippedError('The `contentType` property is not available - `skipNavigation` was used');
|
|
245
|
+
},
|
|
246
|
+
get body() {
|
|
247
|
+
throw new NavigationSkippedError('The `body` property is not available - `skipNavigation` was used');
|
|
248
|
+
},
|
|
249
|
+
get json() {
|
|
250
|
+
throw new NavigationSkippedError('The `json` property is not available - `skipNavigation` was used');
|
|
251
|
+
},
|
|
252
|
+
get waitForSelector() {
|
|
253
|
+
throw new NavigationSkippedError('The `waitForSelector` method is not available - `skipNavigation` was used');
|
|
254
|
+
},
|
|
255
|
+
get parseWithCheerio() {
|
|
256
|
+
throw new NavigationSkippedError('The `parseWithCheerio` method is not available - `skipNavigation` was used');
|
|
257
|
+
},
|
|
232
258
|
};
|
|
233
|
-
|
|
234
|
-
|
|
259
|
+
}
|
|
260
|
+
tryCancel();
|
|
261
|
+
// Before `parseResponse`, which throws for error status codes - a 429 the user opted into treating as an
|
|
262
|
+
// error is still a rate limit the domain should back off from.
|
|
263
|
+
if (crawlingContext.response.status === 429) {
|
|
264
|
+
const retryAfter = crawlingContext.response.headers.get('retry-after');
|
|
265
|
+
if (this.recordDomainRateLimit(crawlingContext.request.url, retryAfter)) {
|
|
266
|
+
// This is the one path that never reads the body, so cancel it to release the connection
|
|
267
|
+
// rather than leaving it to the garbage collector.
|
|
268
|
+
await crawlingContext.response.body?.cancel().catch(() => { });
|
|
269
|
+
throw new RequestThrottledError(`${crawlingContext.request.url} responded with 429.`);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
// Reading the body is still part of the navigation, so it draws from the same shared window: on a server
|
|
273
|
+
// that streams the body slowly the request completes (headers arrive) but the body read would otherwise
|
|
274
|
+
// run unbounded. `extendTimeout` from a post-navigation hook has already pushed this deadline out if asked.
|
|
275
|
+
const remaining = remainingNavigationWindowMillis(crawlingContext, this.#navigationTimeoutMillis);
|
|
276
|
+
if (remaining <= 0) {
|
|
277
|
+
throw new TimeoutError(`Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
278
|
+
}
|
|
279
|
+
const parsed = await addTimeoutToPromise(async () => this.parseResponse(crawlingContext.request, crawlingContext.response), remaining, `Navigation timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
280
|
+
tryCancel();
|
|
281
|
+
const response = parsed.response;
|
|
282
|
+
const contentType = parsed.contentType;
|
|
283
|
+
const loadBody = async () => {
|
|
284
|
+
const { load } = await import('cheerio/slim');
|
|
285
|
+
return load(parsed.body.toString(), { xmlMode: contentType.type.includes('xml') });
|
|
286
|
+
};
|
|
287
|
+
const waitForSelector = async (selector, _timeoutMs) => {
|
|
288
|
+
const $ = await loadBody();
|
|
289
|
+
if ($(selector).get().length === 0) {
|
|
290
|
+
throw new Error(`Selector '${selector}' not found.`);
|
|
291
|
+
}
|
|
292
|
+
};
|
|
293
|
+
const parseWithCheerio = async (selector, timeoutMs) => {
|
|
294
|
+
const $ = await loadBody();
|
|
295
|
+
if (selector) {
|
|
296
|
+
await crawlingContext.waitForSelector(selector, timeoutMs);
|
|
235
297
|
}
|
|
236
|
-
|
|
237
|
-
|
|
298
|
+
return $;
|
|
299
|
+
};
|
|
300
|
+
this.throwOnBlockedRequest(response.status);
|
|
301
|
+
if (this.#saveResponseCookies) {
|
|
302
|
+
try {
|
|
303
|
+
for (const cookie of getCookiesFromResponse(response)) {
|
|
304
|
+
if (!cookie)
|
|
305
|
+
continue;
|
|
306
|
+
try {
|
|
307
|
+
await crawlingContext.session.cookieJar.setCookie(cookie, response.url, {
|
|
308
|
+
ignoreError: false,
|
|
309
|
+
});
|
|
310
|
+
}
|
|
311
|
+
catch (e) {
|
|
312
|
+
this.log.debug(`Could not set cookie: ${e.message}`);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
238
315
|
}
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
this.log.debug(
|
|
242
|
-
// eslint-disable-next-line dot-notation
|
|
243
|
-
`Skipping request ${request.id} (starting url: ${request.url} -> loaded url: ${request.loadedUrl}) because it does not match the enqueue strategy (${request['enqueueStrategy']}).`);
|
|
244
|
-
request.noRetry = true;
|
|
245
|
-
request.state = RequestState.SKIPPED;
|
|
246
|
-
return;
|
|
316
|
+
catch (e) {
|
|
317
|
+
this.log.exception(e, 'Could not get cookies from response');
|
|
247
318
|
}
|
|
248
|
-
Object.assign(crawlingContext, parsed);
|
|
249
|
-
Object.defineProperty(crawlingContext, 'json', {
|
|
250
|
-
get() {
|
|
251
|
-
if (contentType.type !== APPLICATION_JSON_MIME_TYPE)
|
|
252
|
-
return null;
|
|
253
|
-
const jsonString = parsed.body.toString(contentType.encoding);
|
|
254
|
-
return JSON.parse(jsonString);
|
|
255
|
-
},
|
|
256
|
-
});
|
|
257
319
|
}
|
|
320
|
+
return {
|
|
321
|
+
get json() {
|
|
322
|
+
if (contentType.type !== APPLICATION_JSON_MIME_TYPE)
|
|
323
|
+
return null;
|
|
324
|
+
const jsonString = parsed.body.toString(contentType.encoding);
|
|
325
|
+
return JSON.parse(jsonString);
|
|
326
|
+
},
|
|
327
|
+
waitForSelector,
|
|
328
|
+
parseWithCheerio,
|
|
329
|
+
contentType,
|
|
330
|
+
body: parsed.body,
|
|
331
|
+
};
|
|
332
|
+
}
|
|
333
|
+
async handleBlockedRequestByContent(crawlingContext) {
|
|
258
334
|
if (this.retryOnBlocked) {
|
|
259
|
-
const error = await this
|
|
335
|
+
const error = await this.#isRequestBlocked(crawlingContext);
|
|
260
336
|
if (error)
|
|
261
337
|
throw new SessionError(error);
|
|
262
338
|
}
|
|
263
|
-
|
|
264
|
-
try {
|
|
265
|
-
await addTimeoutToPromise(async () => Promise.resolve(this.requestHandler(crawlingContext)), this.userRequestHandlerTimeoutMillis, `requestHandler timed out after ${this.userRequestHandlerTimeoutMillis / 1000} seconds.`);
|
|
266
|
-
request.state = RequestState.DONE;
|
|
267
|
-
}
|
|
268
|
-
catch (e) {
|
|
269
|
-
request.state = RequestState.ERROR;
|
|
270
|
-
throw e;
|
|
271
|
-
}
|
|
339
|
+
return {};
|
|
272
340
|
}
|
|
273
|
-
async isRequestBlocked(crawlingContext) {
|
|
341
|
+
async #isRequestBlocked(crawlingContext) {
|
|
274
342
|
if (HTML_AND_XML_MIME_TYPES.includes(crawlingContext.contentType.type)) {
|
|
275
343
|
const $ = await crawlingContext.parseWithCheerio();
|
|
276
344
|
const foundSelectors = RETRY_CSS_SELECTORS.filter((selector) => $(selector).length > 0);
|
|
@@ -278,87 +346,28 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
278
346
|
return `Found selectors: ${foundSelectors.join(', ')}`;
|
|
279
347
|
}
|
|
280
348
|
}
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
async _handleNavigation(crawlingContext) {
|
|
284
|
-
const gotOptions = {};
|
|
285
|
-
const { request, session } = crawlingContext;
|
|
286
|
-
const preNavigationHooksCookies = this._getCookieHeaderFromRequest(request);
|
|
287
|
-
request.state = RequestState.BEFORE_NAV;
|
|
288
|
-
// Execute pre navigation hooks before applying session pool cookies,
|
|
289
|
-
// as they may also set cookies in the session
|
|
290
|
-
await this._executeHooks(this.preNavigationHooks, crawlingContext, gotOptions);
|
|
291
|
-
tryCancel();
|
|
292
|
-
const postNavigationHooksCookies = this._getCookieHeaderFromRequest(request);
|
|
293
|
-
this._applyCookies(crawlingContext, gotOptions, preNavigationHooksCookies, postNavigationHooksCookies);
|
|
294
|
-
const proxyUrl = crawlingContext.proxyInfo?.url;
|
|
295
|
-
crawlingContext.response = await addTimeoutToPromise(async () => this._requestFunction({ request, session, proxyUrl, gotOptions }), this.navigationTimeoutMillis, `request timed out after ${this.navigationTimeoutMillis / 1000} seconds.`);
|
|
296
|
-
tryCancel();
|
|
297
|
-
request.state = RequestState.AFTER_NAV;
|
|
298
|
-
await this._executeHooks(this.postNavigationHooks, crawlingContext, gotOptions);
|
|
299
|
-
tryCancel();
|
|
300
|
-
}
|
|
301
|
-
/**
|
|
302
|
-
* Sets the cookie header to `gotOptions` based on the provided request and session headers, as well as any changes that occurred due to hooks.
|
|
303
|
-
*/
|
|
304
|
-
_applyCookies({ session, request }, gotOptions, preHookCookies, postHookCookies) {
|
|
305
|
-
const sessionCookie = session?.getCookieString(request.url) ?? '';
|
|
306
|
-
let alteredGotOptionsCookies = gotOptions.headers?.Cookie || gotOptions.headers?.cookie || '';
|
|
307
|
-
if (gotOptions.headers?.Cookie && gotOptions.headers?.cookie) {
|
|
308
|
-
const { Cookie: upperCaseHeader, cookie: lowerCaseHeader } = gotOptions.headers;
|
|
309
|
-
this.log.warning(`Encountered mixed casing for the cookie headers in the got options for request ${request.url} (${request.id}). Their values will be merged`);
|
|
310
|
-
const sourceCookies = [];
|
|
311
|
-
if (Array.isArray(lowerCaseHeader)) {
|
|
312
|
-
sourceCookies.push(...lowerCaseHeader);
|
|
313
|
-
}
|
|
314
|
-
else {
|
|
315
|
-
sourceCookies.push(lowerCaseHeader);
|
|
316
|
-
}
|
|
317
|
-
if (Array.isArray(upperCaseHeader)) {
|
|
318
|
-
sourceCookies.push(...upperCaseHeader);
|
|
319
|
-
}
|
|
320
|
-
else {
|
|
321
|
-
sourceCookies.push(upperCaseHeader);
|
|
322
|
-
}
|
|
323
|
-
alteredGotOptionsCookies = mergeCookies(request.url, sourceCookies);
|
|
324
|
-
}
|
|
325
|
-
const sourceCookies = [sessionCookie, preHookCookies];
|
|
326
|
-
if (Array.isArray(alteredGotOptionsCookies)) {
|
|
327
|
-
sourceCookies.push(...alteredGotOptionsCookies);
|
|
328
|
-
}
|
|
329
|
-
else {
|
|
330
|
-
sourceCookies.push(alteredGotOptionsCookies);
|
|
331
|
-
}
|
|
332
|
-
sourceCookies.push(postHookCookies);
|
|
333
|
-
const mergedCookie = mergeCookies(request.url, sourceCookies);
|
|
334
|
-
gotOptions.headers ??= {};
|
|
335
|
-
Reflect.deleteProperty(gotOptions.headers, 'Cookie');
|
|
336
|
-
Reflect.deleteProperty(gotOptions.headers, 'cookie');
|
|
337
|
-
if (mergedCookie !== '') {
|
|
338
|
-
gotOptions.headers.Cookie = mergedCookie;
|
|
349
|
+
if (this.blockedStatusCodes.has(crawlingContext.response.status)) {
|
|
350
|
+
return `Blocked by status code ${crawlingContext.response.status}`;
|
|
339
351
|
}
|
|
352
|
+
return false;
|
|
340
353
|
}
|
|
341
354
|
/**
|
|
342
355
|
* Function to make the HTTP request. It performs optimizations
|
|
343
356
|
* on the request such as only downloading the request body if the
|
|
344
357
|
* received content type matches text/html, application/xml, application/xhtml+xml.
|
|
345
358
|
*/
|
|
346
|
-
async
|
|
347
|
-
|
|
348
|
-
// @ts-ignore
|
|
349
|
-
({ TimeoutError } = await import('got-scraping'));
|
|
350
|
-
}
|
|
351
|
-
const opts = this._getRequestOptions(request, session, proxyUrl, gotOptions);
|
|
359
|
+
async requestFunction({ request, session, proxyUrl }) {
|
|
360
|
+
const opts = this.getRequestOptions(request, proxyUrl);
|
|
352
361
|
try {
|
|
353
|
-
return await this.
|
|
362
|
+
return await this.requestAsBrowser(opts, session);
|
|
354
363
|
}
|
|
355
364
|
catch (e) {
|
|
356
|
-
if (e instanceof TimeoutError) {
|
|
357
|
-
this.
|
|
358
|
-
return
|
|
365
|
+
if (e instanceof Error && e.constructor.name === 'TimeoutError') {
|
|
366
|
+
this.handleRequestTimeout(session);
|
|
367
|
+
return new Response(); // this will never happen, as handleRequestTimeout always throws
|
|
359
368
|
}
|
|
360
369
|
if (this.isProxyError(e)) {
|
|
361
|
-
throw new SessionError(this.
|
|
370
|
+
throw new SessionError(this.getMessageFromError(e));
|
|
362
371
|
}
|
|
363
372
|
else {
|
|
364
373
|
throw e;
|
|
@@ -368,18 +377,15 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
368
377
|
/**
|
|
369
378
|
* Encodes and parses response according to the provided content type
|
|
370
379
|
*/
|
|
371
|
-
async
|
|
372
|
-
const {
|
|
373
|
-
const { type, charset } = parseContentTypeFromResponse(
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
if (statusCode >= 400 && statusCode <= 599) {
|
|
377
|
-
this.stats.registerStatusCode(statusCode);
|
|
380
|
+
async parseResponse(request, response) {
|
|
381
|
+
const { status } = response;
|
|
382
|
+
const { type, charset } = parseContentTypeFromResponse(response);
|
|
383
|
+
if (status >= 400 && status <= 599) {
|
|
384
|
+
this.statistics.registerStatusCode(status);
|
|
378
385
|
}
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
const body = await readStreamToString(response, encoding);
|
|
386
|
+
if (this.isErrorStatusCode(status)) {
|
|
387
|
+
const { response: decoded } = this.encodeResponse(request, response, charset);
|
|
388
|
+
const body = await decoded.text(); // TODO - this always uses UTF-8 (see https://developer.mozilla.org/en-US/docs/Web/API/Request/text)
|
|
383
389
|
// Errors are often sent as JSON, so attempt to parse them,
|
|
384
390
|
// despite Accept header being set to text/html.
|
|
385
391
|
if (type === APPLICATION_JSON_MIME_TYPE) {
|
|
@@ -387,74 +393,62 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
387
393
|
let { message } = errorResponse;
|
|
388
394
|
if (!message)
|
|
389
395
|
message = util.inspect(errorResponse, { depth: 1, maxArrayLength: 10 });
|
|
390
|
-
throw new Error(`${
|
|
396
|
+
throw new Error(`${status} - ${message}`);
|
|
391
397
|
}
|
|
392
|
-
if (
|
|
393
|
-
throw new Error(`${
|
|
398
|
+
if (this.additionalHttpErrorStatusCodes.has(status)) {
|
|
399
|
+
throw new Error(`${status} - Error status code was set by user.`);
|
|
394
400
|
}
|
|
395
401
|
// It's not a JSON, so it's probably some text. Get the first 100 chars of it.
|
|
396
|
-
throw new Error(`${
|
|
402
|
+
throw new Error(`${status} - Internal Server Error: ${body.slice(0, 100)}`);
|
|
397
403
|
}
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
404
|
+
if (HTML_AND_XML_MIME_TYPES.includes(type) && !charset && !this.#forceResponseEncoding) {
|
|
405
|
+
// The charset comes from the document itself, so the raw bytes are what we need -
|
|
406
|
+
// decoding them through `encodeResponse` first would consume the body for nothing.
|
|
407
|
+
const rawBytes = Buffer.from(await response.arrayBuffer());
|
|
408
|
+
const metaCharset = extractCharsetFromHtmlBytes(rawBytes);
|
|
409
|
+
const charsetToUse = metaCharset ?? this.#suggestResponseEncoding ?? 'utf-8';
|
|
410
|
+
const body = iconv.encodingExists(charsetToUse)
|
|
411
|
+
? iconv.decode(rawBytes, charsetToUse)
|
|
412
|
+
: rawBytes.toString('utf8');
|
|
413
|
+
return { response, contentType: { type, encoding: 'utf-8' }, body };
|
|
402
414
|
}
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
response,
|
|
408
|
-
contentType,
|
|
409
|
-
enqueueLinks: async () => Promise.resolve({ processedRequests: [], unprocessedRequests: [] }),
|
|
410
|
-
};
|
|
415
|
+
const { response: reencodedResponse, encoding } = this.encodeResponse(request, response, charset);
|
|
416
|
+
const contentType = { type, encoding };
|
|
417
|
+
if (HTML_AND_XML_MIME_TYPES.includes(type)) {
|
|
418
|
+
return { response, contentType, body: await reencodedResponse.text() };
|
|
411
419
|
}
|
|
412
|
-
}
|
|
413
|
-
async _parseHTML(response, _isXml, _crawlingContext) {
|
|
414
420
|
return {
|
|
415
|
-
body: await
|
|
421
|
+
body: Buffer.from(await reencodedResponse.bytes()),
|
|
422
|
+
response,
|
|
423
|
+
contentType,
|
|
416
424
|
};
|
|
417
425
|
}
|
|
418
426
|
/**
|
|
419
427
|
* Combines the provided `requestOptions` with mandatory (non-overridable) values.
|
|
420
428
|
*/
|
|
421
|
-
|
|
429
|
+
getRequestOptions(request, proxyUrl) {
|
|
422
430
|
const requestOptions = {
|
|
423
431
|
url: request.url,
|
|
424
432
|
method: request.method,
|
|
425
433
|
proxyUrl,
|
|
426
|
-
timeout:
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
headers: { ...request.headers, ...gotOptions?.headers },
|
|
430
|
-
https: {
|
|
431
|
-
...gotOptions?.https,
|
|
432
|
-
rejectUnauthorized: !this.ignoreSslErrors,
|
|
433
|
-
},
|
|
434
|
-
isStream: true,
|
|
434
|
+
timeout: this.#navigationTimeoutMillis,
|
|
435
|
+
headers: request.headers,
|
|
436
|
+
body: undefined,
|
|
435
437
|
};
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
// on individual proxy level, not on the `proxyConfiguration` level,
|
|
440
|
-
// because users can use normal + MITM proxies in a single configuration.
|
|
441
|
-
// Disable SSL verification for MITM proxies
|
|
442
|
-
if (this.proxyConfiguration && this.proxyConfiguration.isManInTheMiddle) {
|
|
443
|
-
requestOptions.https = {
|
|
444
|
-
...requestOptions.https,
|
|
445
|
-
rejectUnauthorized: false,
|
|
446
|
-
};
|
|
438
|
+
if (requestOptions.headers?.cookie || requestOptions.headers?.Cookie) {
|
|
439
|
+
requestOptions.headers.Cookie = this.getCookieHeaderFromRequest(request);
|
|
440
|
+
delete requestOptions.headers.cookie;
|
|
447
441
|
}
|
|
448
442
|
if (/PATCH|POST|PUT/.test(request.method))
|
|
449
443
|
requestOptions.body = request.payload ?? '';
|
|
450
444
|
return requestOptions;
|
|
451
445
|
}
|
|
452
|
-
|
|
453
|
-
if (this
|
|
454
|
-
encoding = this
|
|
446
|
+
encodeResponse(request, response, encoding) {
|
|
447
|
+
if (this.#forceResponseEncoding) {
|
|
448
|
+
encoding = this.#forceResponseEncoding;
|
|
455
449
|
}
|
|
456
|
-
else if (!encoding && this
|
|
457
|
-
encoding = this
|
|
450
|
+
else if (!encoding && this.#suggestResponseEncoding) {
|
|
451
|
+
encoding = this.#suggestResponseEncoding;
|
|
458
452
|
}
|
|
459
453
|
// Fall back to utf-8 if we still don't have encoding.
|
|
460
454
|
const utf8 = 'utf8';
|
|
@@ -467,14 +461,16 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
467
461
|
// Try to re-encode a variety of unsupported encodings to utf-8
|
|
468
462
|
if (iconv.encodingExists(encoding)) {
|
|
469
463
|
const encodeStream = iconv.encodeStream(utf8);
|
|
470
|
-
const decodeStream = iconv
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
464
|
+
const decodeStream = iconv
|
|
465
|
+
.decodeStream(encoding)
|
|
466
|
+
.on('error', (err) => encodeStream.emit('error', err));
|
|
467
|
+
const reencodedBody = response.body
|
|
468
|
+
? Readable.toWeb(Readable.from(Readable.fromWeb(response.body)
|
|
469
|
+
.pipe(decodeStream)
|
|
470
|
+
.pipe(encodeStream)))
|
|
471
|
+
: null;
|
|
476
472
|
return {
|
|
477
|
-
response:
|
|
473
|
+
response: new ResponseWithUrl(reencodedBody, response),
|
|
478
474
|
encoding: utf8,
|
|
479
475
|
};
|
|
480
476
|
}
|
|
@@ -483,15 +479,15 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
483
479
|
/**
|
|
484
480
|
* Checks and extends supported mime types
|
|
485
481
|
*/
|
|
486
|
-
|
|
482
|
+
extendSupportedMimeTypes(additionalMimeTypes) {
|
|
487
483
|
for (const mimeType of additionalMimeTypes) {
|
|
488
484
|
if (mimeType === '*/*') {
|
|
489
|
-
this
|
|
485
|
+
this.#supportedMimeTypes.add(mimeType);
|
|
490
486
|
continue;
|
|
491
487
|
}
|
|
492
488
|
try {
|
|
493
489
|
const parsedType = contentTypeParser.parse(mimeType);
|
|
494
|
-
this
|
|
490
|
+
this.#supportedMimeTypes.add(parsedType.type);
|
|
495
491
|
}
|
|
496
492
|
catch (err) {
|
|
497
493
|
throw new Error(`Can not parse mime type ${mimeType} from "options.additionalMimeTypes".`);
|
|
@@ -501,136 +497,55 @@ export class HttpCrawler extends BasicCrawler {
|
|
|
501
497
|
/**
|
|
502
498
|
* Handles timeout request
|
|
503
499
|
*/
|
|
504
|
-
|
|
505
|
-
session
|
|
506
|
-
throw new Error(`
|
|
500
|
+
handleRequestTimeout(session) {
|
|
501
|
+
session.markBad();
|
|
502
|
+
throw new Error(`Request timed out after ${this.#navigationTimeoutMillis / 1000} seconds.`);
|
|
507
503
|
}
|
|
508
|
-
|
|
509
|
-
const {
|
|
504
|
+
abortDownloadOfBody(request, response) {
|
|
505
|
+
const { status } = response;
|
|
510
506
|
const { type } = parseContentTypeFromResponse(response);
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
// if we retry the request, can the Content-Type change?
|
|
514
|
-
const isTransientContentType = statusCode >= 500 || blockedStatusCodes.includes(statusCode);
|
|
515
|
-
if (!this.supportedMimeTypes.has(type) && !this.supportedMimeTypes.has('*/*') && !isTransientContentType) {
|
|
507
|
+
const isTransientContentType = status >= 500 || this.blockedStatusCodes.has(status);
|
|
508
|
+
if (!this.#supportedMimeTypes.has(type) && !this.#supportedMimeTypes.has('*/*') && !isTransientContentType) {
|
|
516
509
|
request.noRetry = true;
|
|
517
510
|
throw new Error(`Resource ${request.url} served Content-Type ${type}, ` +
|
|
518
|
-
`but only ${Array.from(this
|
|
511
|
+
`but only ${Array.from(this.#supportedMimeTypes).join(', ')} are allowed. Skipping resource.`);
|
|
519
512
|
}
|
|
520
513
|
}
|
|
521
514
|
/**
|
|
522
515
|
* @internal wraps public utility for mocking purposes
|
|
523
516
|
*/
|
|
524
|
-
|
|
525
|
-
const
|
|
517
|
+
requestAsBrowser = async (options, session) => {
|
|
518
|
+
const opts = processHttpRequestOptions({
|
|
526
519
|
...options,
|
|
527
|
-
cookieJar: options.cookieJar, // HACK - the type of ToughCookieJar in got is wrong
|
|
528
520
|
responseType: 'text',
|
|
529
|
-
}), (redirectResponse, updatedRequest) => {
|
|
530
|
-
if (this.persistCookiesPerSession) {
|
|
531
|
-
session.setCookiesFromResponse(redirectResponse);
|
|
532
|
-
const cookieString = session.getCookieString(updatedRequest.url.toString());
|
|
533
|
-
if (cookieString !== '') {
|
|
534
|
-
updatedRequest.headers.Cookie = cookieString;
|
|
535
|
-
}
|
|
536
|
-
}
|
|
537
521
|
});
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
// @ts-expect-error
|
|
564
|
-
if (stream.rawTrailers)
|
|
565
|
-
stream.rawTrailers = response.rawTrailers; // TODO BC with got - remove in 4.0
|
|
566
|
-
// @ts-expect-error
|
|
567
|
-
if (stream.trailers)
|
|
568
|
-
stream.trailers = response.trailers;
|
|
569
|
-
// @ts-expect-error
|
|
570
|
-
stream.complete = response.complete;
|
|
571
|
-
});
|
|
572
|
-
for (const prop of properties) {
|
|
573
|
-
if (!(prop in stream)) {
|
|
574
|
-
stream[prop] = response[prop];
|
|
575
|
-
}
|
|
576
|
-
}
|
|
577
|
-
return stream;
|
|
578
|
-
}
|
|
579
|
-
/**
|
|
580
|
-
* Gets parsed content type from response object
|
|
581
|
-
* @param response HTTP response object
|
|
582
|
-
*/
|
|
583
|
-
function parseContentTypeFromResponse(response) {
|
|
584
|
-
ow(response, ow.object.partialShape({
|
|
585
|
-
url: ow.string.url,
|
|
586
|
-
headers: new ObjectPredicate(),
|
|
587
|
-
}));
|
|
588
|
-
const { url, headers } = response;
|
|
589
|
-
let parsedContentType;
|
|
590
|
-
if (headers['content-type']) {
|
|
591
|
-
try {
|
|
592
|
-
parsedContentType = contentTypeParser.parse(headers['content-type']);
|
|
593
|
-
}
|
|
594
|
-
catch {
|
|
595
|
-
// Can not parse content type from Content-Type header. Try to parse it from file extension.
|
|
596
|
-
}
|
|
597
|
-
}
|
|
598
|
-
// Parse content type from file extension as fallback
|
|
599
|
-
if (!parsedContentType) {
|
|
600
|
-
const parsedUrl = new URL(url);
|
|
601
|
-
const contentTypeFromExtname = mime.contentType(extname(parsedUrl.pathname)) || 'application/octet-stream; charset=utf-8'; // Fallback content type, specified in https://tools.ietf.org/html/rfc7231#section-3.1.1.5
|
|
602
|
-
parsedContentType = contentTypeParser.parse(contentTypeFromExtname);
|
|
603
|
-
}
|
|
604
|
-
return {
|
|
605
|
-
type: parsedContentType.type,
|
|
606
|
-
charset: parsedContentType.parameters.charset,
|
|
522
|
+
// When saveResponseCookies is false, the response cookies must not mutate the
|
|
523
|
+
// session jar. Reads still go through the session (so session.setCookie() in pre-nav
|
|
524
|
+
// hooks keeps working) but a per-request clone is passed in so writes are discarded.
|
|
525
|
+
const cookieJar = this.#saveResponseCookies ? session.cookieJar : await session.cookieJar.clone();
|
|
526
|
+
// Bind the request to the shared navigation window instead of a fixed per-request timeout, so
|
|
527
|
+
// `extendTimeout()` can push the deadline and a fixed `AbortSignal.timeout` won't fire on its own and
|
|
528
|
+
// kill a lazily-read body mid-extension. This aborts the socket only during the header phase; the body
|
|
529
|
+
// read is bounded separately at the promise level (see `processHttpResponse`), so a slow-streaming body
|
|
530
|
+
// still fails cleanly with a navigation timeout, though the socket is left to close on its own.
|
|
531
|
+
const cancelSignal = storage.getStore()?.cancelTask.signal;
|
|
532
|
+
const response = await this.httpClient.sendRequest(new Request(opts.url, {
|
|
533
|
+
body: opts.body ? Readable.toWeb(opts.body) : undefined,
|
|
534
|
+
headers: new Headers(opts.headers),
|
|
535
|
+
method: opts.method,
|
|
536
|
+
// Node-specific option to make the request body work with streams
|
|
537
|
+
duplex: 'half',
|
|
538
|
+
}), {
|
|
539
|
+
session,
|
|
540
|
+
cookieJar,
|
|
541
|
+
proxyUrl: opts.proxyUrl,
|
|
542
|
+
signal: cancelSignal,
|
|
543
|
+
timeoutMillis: cancelSignal ? undefined : opts.timeout,
|
|
544
|
+
ignoreTlsErrors: this.#ignoreTlsErrors,
|
|
545
|
+
});
|
|
546
|
+
return response;
|
|
607
547
|
};
|
|
608
548
|
}
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
* This instance can then serve as a `requestHandler` of your {@link HttpCrawler}.
|
|
612
|
-
* Defaults to the {@link HttpCrawlingContext}.
|
|
613
|
-
*
|
|
614
|
-
* > Serves as a shortcut for using `Router.create<HttpCrawlingContext>()`.
|
|
615
|
-
*
|
|
616
|
-
* ```ts
|
|
617
|
-
* import { HttpCrawler, createHttpRouter } from 'crawlee';
|
|
618
|
-
*
|
|
619
|
-
* const router = createHttpRouter();
|
|
620
|
-
* router.addHandler('label-a', async (ctx) => {
|
|
621
|
-
* ctx.log.info('...');
|
|
622
|
-
* });
|
|
623
|
-
* router.addDefaultHandler(async (ctx) => {
|
|
624
|
-
* ctx.log.info('...');
|
|
625
|
-
* });
|
|
626
|
-
*
|
|
627
|
-
* const crawler = new HttpCrawler({
|
|
628
|
-
* requestHandler: router,
|
|
629
|
-
* });
|
|
630
|
-
* await crawler.run();
|
|
631
|
-
* ```
|
|
632
|
-
*/
|
|
633
|
-
export function createHttpRouter(routes) {
|
|
634
|
-
return Router.create(routes);
|
|
549
|
+
export function createHttpRouter(routesOrSchemas) {
|
|
550
|
+
return Router.create(routesOrSchemas);
|
|
635
551
|
}
|
|
636
|
-
//# sourceMappingURL=http-crawler.js.map
|