@crawlee/http 4.0.0-beta.2 → 4.0.0-beta.200
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/index.d.ts +1 -1
- package/index.js +1 -1
- package/internals/dom-crawler.d.ts +131 -0
- package/internals/dom-crawler.js +93 -0
- package/internals/file-download.d.ts +23 -46
- package/internals/file-download.js +53 -109
- package/internals/http-crawler.d.ts +226 -293
- package/internals/http-crawler.js +344 -429
- package/internals/utils.d.ts +19 -0
- package/internals/utils.js +79 -0
- package/package.json +13 -12
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/file-download.d.ts.map +0 -1
- package/internals/file-download.js.map +0 -1
- package/internals/http-crawler.d.ts.map +0 -1
- package/internals/http-crawler.js.map +0 -1
- package/tsconfig.build.tsbuildinfo +0 -1
|
@@ -1,72 +1,87 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
import
|
|
3
|
-
import type { BasicCrawlerOptions, CrawlingContext, ErrorHandler, GetUserDataFromRequest, ProxyConfiguration, Request, RequestHandler, RouterRoutes, Session } from '@crawlee/basic';
|
|
4
|
-
import { BasicCrawler, Configuration, CrawlerExtension } from '@crawlee/basic';
|
|
5
|
-
import type { HttpResponse } from '@crawlee/core';
|
|
1
|
+
import type { BasicCrawlerOptions, ConcurrencySystem, ConcurrencySystemOptions, CrawlingContext, ErrorHandler, GetUserDataFromRequest, LoadedRequest, CrawlingRequest, RequireContextPipeline, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/basic';
|
|
2
|
+
import { BasicCrawler, ContextPipeline } from '@crawlee/basic';
|
|
6
3
|
import type { Awaitable, Dictionary } from '@crawlee/types';
|
|
7
|
-
import {
|
|
8
|
-
import type { RequestLike, ResponseLike } from 'content-type';
|
|
9
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
10
|
-
import type { Method, OptionsInit } from 'got-scraping';
|
|
11
|
-
import { ObjectPredicate } from 'ow';
|
|
4
|
+
import type { CheerioAPI } from 'cheerio';
|
|
12
5
|
import type { JsonValue } from 'type-fest';
|
|
6
|
+
import { z } from 'zod';
|
|
13
7
|
/**
|
|
14
|
-
*
|
|
15
|
-
* @
|
|
8
|
+
* A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event
|
|
9
|
+
* loop. {@link HttpCrawler} folds these into the {@link ConcurrencySystem} it builds by default, with your own
|
|
10
|
+
* concurrency shortcuts (`minConcurrency`, `maxConcurrency`, `maxRequestsPerMinute`) kept on top.
|
|
11
|
+
*
|
|
12
|
+
* A {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default
|
|
13
|
+
* wholesale, tuning included, so spread these options in if you want to keep it:
|
|
14
|
+
*
|
|
15
|
+
* ```typescript
|
|
16
|
+
* new ConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, maxConcurrency: 50 });
|
|
17
|
+
* ```
|
|
16
18
|
*/
|
|
17
|
-
export
|
|
18
|
-
body?: unknown;
|
|
19
|
-
};
|
|
19
|
+
export declare const HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS: ConcurrencySystemOptions;
|
|
20
20
|
export type HttpErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
21
|
-
JSONData extends JsonValue = any
|
|
22
|
-
|
|
21
|
+
JSONData extends JsonValue = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
22
|
+
ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, HttpCrawlingContext<UserData, JSONData> & ContextExtension>;
|
|
23
|
+
export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
23
24
|
/**
|
|
24
|
-
* Timeout
|
|
25
|
+
* Timeout for the whole navigation phase, given in seconds. A single window shared by the
|
|
26
|
+
* `preNavigationHooks`, the navigation (the HTTP request to the resource), and the `postNavigationHooks` -
|
|
27
|
+
* so a slow hook eats into the same budget the navigation uses. Separate from the
|
|
28
|
+
* {@link BasicCrawlerOptions.requestHandlerTimeoutSecs|`requestHandlerTimeoutSecs`}, which times only the
|
|
29
|
+
* request handler.
|
|
25
30
|
*/
|
|
26
31
|
navigationTimeoutSecs?: number;
|
|
27
32
|
/**
|
|
28
|
-
* If set to true
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
* For more information, see the [documentation](https://docs.apify.com/proxy).
|
|
33
|
+
* If set to `true`, TLS/SSL certificate errors are ignored. Forwarded to the HTTP client as
|
|
34
|
+
* {@link SendRequestOptions.ignoreTlsErrors|`ignoreTlsErrors`} on every navigation request, so custom
|
|
35
|
+
* {@link BaseHttpClient} implementations should honor that flag (the built-in impit and got-scraping
|
|
36
|
+
* clients do; the native fetch fallback cannot disable TLS verification and warns instead).
|
|
37
|
+
*
|
|
38
|
+
* @default true
|
|
35
39
|
*/
|
|
36
|
-
|
|
40
|
+
ignoreTlsErrors?: boolean;
|
|
37
41
|
/**
|
|
38
42
|
* Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
|
|
39
|
-
* or browser properties before navigation. The function accepts
|
|
40
|
-
* which
|
|
43
|
+
* or browser properties before navigation. The function accepts one parameter `crawlingContext`,
|
|
44
|
+
* which is passed to the `requestAsBrowser()` function the crawler calls to navigate.
|
|
45
|
+
*
|
|
46
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context,
|
|
47
|
+
* allowing the hook to override context members for subsequent hooks and pipeline stages.
|
|
48
|
+
*
|
|
49
|
+
* The context is built up in the following order: base context (`request`, `session`, helpers, ...) ->
|
|
50
|
+
* `extendContext` -> `preNavigationHooks` -> navigation -> `postNavigationHooks` -> `requestHandler`.
|
|
51
|
+
* This means the members added by `extendContext` are already available here, but navigation-dependent
|
|
52
|
+
* members (e.g. `response`, `body`, `$`) are not.
|
|
41
53
|
* Example:
|
|
42
54
|
* ```
|
|
43
55
|
* preNavigationHooks: [
|
|
44
|
-
* async (crawlingContext
|
|
56
|
+
* async (crawlingContext) => {
|
|
45
57
|
* // ...
|
|
46
58
|
* },
|
|
47
59
|
* ]
|
|
48
60
|
* ```
|
|
49
|
-
*
|
|
50
|
-
* Modyfing `pageOptions` is supported only in Playwright incognito.
|
|
51
|
-
* See {@link PrePageCreateHook}
|
|
52
61
|
*/
|
|
53
|
-
preNavigationHooks?: InternalHttpHook<
|
|
62
|
+
preNavigationHooks?: InternalHttpHook<CrawlingContext<any>, ContextExtension>[];
|
|
54
63
|
/**
|
|
55
64
|
* Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
|
|
56
65
|
* The function accepts `crawlingContext` as the only parameter.
|
|
66
|
+
*
|
|
67
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context,
|
|
68
|
+
* which is useful for overriding the `response` after solving a challenge or re-fetching the resource.
|
|
57
69
|
* Example:
|
|
58
70
|
* ```
|
|
59
71
|
* postNavigationHooks: [
|
|
60
72
|
* async (crawlingContext) => {
|
|
61
|
-
*
|
|
73
|
+
* if (await needsRevalidation(crawlingContext)) {
|
|
74
|
+
* return { response: await refetch(crawlingContext.request) };
|
|
75
|
+
* }
|
|
62
76
|
* },
|
|
63
77
|
* ]
|
|
64
78
|
* ```
|
|
65
79
|
*/
|
|
66
|
-
postNavigationHooks?: InternalHttpHook<
|
|
80
|
+
postNavigationHooks?: InternalHttpHook<CrawlingContextWithResponse, ContextExtension>[];
|
|
67
81
|
/**
|
|
68
82
|
* An array of [MIME types](https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Complete_list_of_MIME_types)
|
|
69
|
-
* you want the crawler to load and process. By default, only `text/html
|
|
83
|
+
* you want the crawler to load and process. By default, only `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
84
|
+
* and `application/json` MIME types are supported.
|
|
70
85
|
*/
|
|
71
86
|
additionalMimeTypes?: string[];
|
|
72
87
|
/**
|
|
@@ -92,35 +107,26 @@ export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext
|
|
|
92
107
|
*/
|
|
93
108
|
forceResponseEncoding?: string;
|
|
94
109
|
/**
|
|
95
|
-
* Automatically saves cookies to Session.
|
|
110
|
+
* Automatically saves cookies to Session. Enabled by default.
|
|
96
111
|
*
|
|
97
112
|
* It parses cookie from response "set-cookie" header saves or updates cookies for session and once the session is used for next request.
|
|
98
113
|
* It passes the "Cookie" header to the request with the session cookies.
|
|
99
114
|
*/
|
|
100
|
-
|
|
115
|
+
saveResponseCookies?: boolean;
|
|
116
|
+
}
|
|
117
|
+
export type InternalHttpHook<Context, ContextExtension = {}> = (crawlingContext: Context & ContextExtension) => Awaitable<void | Partial<Context>>;
|
|
118
|
+
interface CrawlingContextWithResponse<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
|
|
101
119
|
/**
|
|
102
|
-
*
|
|
103
|
-
* By default, status codes >= 500 trigger errors.
|
|
120
|
+
* The request object that was successfully loaded and navigated to, including the {@link Request.loadedUrl|`loadedUrl`} property.
|
|
104
121
|
*/
|
|
105
|
-
|
|
122
|
+
request: LoadedRequest<CrawlingRequest<UserData>>;
|
|
106
123
|
/**
|
|
107
|
-
*
|
|
108
|
-
* By default, status codes >= 500 trigger errors.
|
|
124
|
+
* The HTTP response object containing status code, headers, and other response metadata.
|
|
109
125
|
*/
|
|
110
|
-
|
|
126
|
+
response: Response;
|
|
111
127
|
}
|
|
112
|
-
/**
|
|
113
|
-
* @internal
|
|
114
|
-
*/
|
|
115
|
-
export type InternalHttpHook<Context> = (crawlingContext: Context, gotOptions: OptionsInit) => Awaitable<void>;
|
|
116
|
-
export type HttpHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
117
|
-
JSONData extends JsonValue = any> = InternalHttpHook<HttpCrawlingContext<UserData, JSONData>>;
|
|
118
|
-
/**
|
|
119
|
-
* @internal
|
|
120
|
-
*/
|
|
121
128
|
export interface InternalHttpCrawlingContext<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
122
|
-
JSONData extends JsonValue = any
|
|
123
|
-
Crawler = HttpCrawler<any>> extends CrawlingContext<Crawler, UserData> {
|
|
129
|
+
JSONData extends JsonValue = any> extends CrawlingContextWithResponse<UserData> {
|
|
124
130
|
/**
|
|
125
131
|
* The request body of the web page.
|
|
126
132
|
* The type depends on the `Content-Type` header of the web page:
|
|
@@ -139,9 +145,12 @@ Crawler = HttpCrawler<any>> extends CrawlingContext<Crawler, UserData> {
|
|
|
139
145
|
type: string;
|
|
140
146
|
encoding: BufferEncoding;
|
|
141
147
|
};
|
|
142
|
-
response: PlainResponse;
|
|
143
148
|
/**
|
|
144
|
-
* Wait for an element matching the selector to appear.
|
|
149
|
+
* Wait for an element matching the selector to appear.
|
|
150
|
+
*
|
|
151
|
+
* `HttpCrawler` and {@link CheerioCrawler} parse a response that is already fully downloaded, so there is
|
|
152
|
+
* nothing to wait for and `timeoutMs` is ignored. {@link JSDOMCrawler} and {@link LinkeDOMCrawler} poll for
|
|
153
|
+
* the selector instead, with `timeoutMs` defaulting to 5s.
|
|
145
154
|
*
|
|
146
155
|
* **Example usage:**
|
|
147
156
|
* ```ts
|
|
@@ -155,7 +164,12 @@ Crawler = HttpCrawler<any>> extends CrawlingContext<Crawler, UserData> {
|
|
|
155
164
|
waitForSelector(selector: string, timeoutMs?: number): Promise<void>;
|
|
156
165
|
/**
|
|
157
166
|
* Returns Cheerio handle for `page.content()`, allowing to work with the data same way as with {@link CheerioCrawler}.
|
|
158
|
-
*
|
|
167
|
+
* This is here to unify the crawler API, so they all have this handy method - in {@link CheerioCrawler} it has
|
|
168
|
+
* the same return type as the `$` context property, so use it only if you are abstracting your workflow to
|
|
169
|
+
* support different context types in one handler.
|
|
170
|
+
*
|
|
171
|
+
* When provided with the `selector` argument, it will throw if it's not available. {@link JSDOMCrawler} and
|
|
172
|
+
* {@link LinkeDOMCrawler} wait for the selector first, with `timeoutMs` defaulting to 5s.
|
|
159
173
|
*
|
|
160
174
|
* **Example usage:**
|
|
161
175
|
* ```ts
|
|
@@ -165,9 +179,9 @@ Crawler = HttpCrawler<any>> extends CrawlingContext<Crawler, UserData> {
|
|
|
165
179
|
* });
|
|
166
180
|
* ```
|
|
167
181
|
*/
|
|
168
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<
|
|
182
|
+
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
169
183
|
}
|
|
170
|
-
export interface HttpCrawlingContext<UserData extends Dictionary = any, JSONData extends JsonValue = any> extends InternalHttpCrawlingContext<UserData, JSONData
|
|
184
|
+
export interface HttpCrawlingContext<UserData extends Dictionary = any, JSONData extends JsonValue = any> extends InternalHttpCrawlingContext<UserData, JSONData> {
|
|
171
185
|
}
|
|
172
186
|
export type HttpRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
173
187
|
JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData, JSONData>>;
|
|
@@ -182,38 +196,40 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
|
|
|
182
196
|
*
|
|
183
197
|
* This crawler downloads each URL using a plain HTTP request and doesn't do any HTML parsing.
|
|
184
198
|
*
|
|
185
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
186
|
-
* {@link
|
|
187
|
-
*
|
|
199
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
200
|
+
* {@link IRequestManager|request manager} provided via the {@link HttpCrawlerOptions.requestManager|`requestManager`}
|
|
201
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
202
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
203
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
204
|
+
* result as `requestManager`.
|
|
188
205
|
*
|
|
189
|
-
*
|
|
190
|
-
*
|
|
191
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
206
|
+
* > The {@link HttpCrawlerOptions.requestList|`requestList`} and {@link HttpCrawlerOptions.requestQueue|`requestQueue`}
|
|
207
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
192
208
|
*
|
|
193
209
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
194
210
|
*
|
|
195
|
-
* We can use the `preNavigationHooks` to adjust
|
|
211
|
+
* We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
|
|
196
212
|
*
|
|
197
213
|
* ```javascript
|
|
198
214
|
* preNavigationHooks: [
|
|
199
|
-
* (crawlingContext
|
|
215
|
+
* (crawlingContext) => {
|
|
200
216
|
* // ...
|
|
201
217
|
* },
|
|
202
218
|
* ]
|
|
203
219
|
* ```
|
|
204
220
|
*
|
|
205
|
-
* By default, this crawler only processes web pages with the `text/html`
|
|
206
|
-
* and `application/
|
|
221
|
+
* By default, this crawler only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
222
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
207
223
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
208
224
|
* use the {@link HttpCrawlerOptions.additionalMimeTypes} constructor option.
|
|
209
225
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
210
226
|
* For details, see {@link HttpCrawlerOptions.requestHandler}.
|
|
211
227
|
*
|
|
212
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
213
|
-
*
|
|
214
|
-
*
|
|
215
|
-
*
|
|
216
|
-
* {@link
|
|
228
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
229
|
+
* {@link ConcurrencySystem}.
|
|
230
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
231
|
+
* constructor, or, for finer control, by injecting a pre-configured
|
|
232
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
217
233
|
*
|
|
218
234
|
* **Example usage:**
|
|
219
235
|
*
|
|
@@ -238,248 +254,165 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
|
|
|
238
254
|
* ```
|
|
239
255
|
* @category Crawlers
|
|
240
256
|
*/
|
|
241
|
-
export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any,
|
|
242
|
-
|
|
257
|
+
export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any> = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawler<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
258
|
+
#private;
|
|
243
259
|
/**
|
|
244
|
-
*
|
|
245
|
-
* Only available if used by the crawler.
|
|
260
|
+
* @internal
|
|
246
261
|
*/
|
|
247
|
-
proxyConfiguration?: ProxyConfiguration;
|
|
248
|
-
protected userRequestHandlerTimeoutMillis: number;
|
|
249
|
-
protected preNavigationHooks: InternalHttpHook<Context>[];
|
|
250
|
-
protected postNavigationHooks: InternalHttpHook<Context>[];
|
|
251
|
-
protected persistCookiesPerSession: boolean;
|
|
252
|
-
protected navigationTimeoutMillis: number;
|
|
253
|
-
protected ignoreSslErrors: boolean;
|
|
254
|
-
protected suggestResponseEncoding?: string;
|
|
255
|
-
protected forceResponseEncoding?: string;
|
|
256
|
-
protected additionalHttpErrorStatusCodes: Set<number>;
|
|
257
|
-
protected ignoreHttpErrorStatusCodes: Set<number>;
|
|
258
|
-
protected readonly supportedMimeTypes: Set<string>;
|
|
259
262
|
protected static optionsShape: {
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
315
|
-
respectRobotsTxtFile: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
316
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
317
|
-
onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
|
|
318
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
319
|
-
httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
320
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
321
|
-
minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
322
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
323
|
-
maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
324
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
325
|
-
maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
326
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
327
|
-
keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
328
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
329
|
-
log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
330
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
331
|
-
experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
332
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
333
|
-
statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
263
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
264
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
265
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
266
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
267
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
268
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
269
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
270
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
271
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
272
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
273
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
274
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
275
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
276
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
277
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
278
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
279
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
280
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
281
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
282
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
283
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
284
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
285
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
286
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
287
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
288
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
289
|
+
deferred: "deferred";
|
|
290
|
+
writeThrough: "writeThrough";
|
|
291
|
+
}>>;
|
|
292
|
+
}, z.core.$strict>]>>;
|
|
293
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
294
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
295
|
+
httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
|
|
296
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
297
|
+
configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
|
|
298
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
299
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
300
|
+
eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
|
|
301
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
302
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
303
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
304
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
305
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
306
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
307
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
308
|
+
id: z.ZodOptional<z.ZodString>;
|
|
309
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
310
|
+
ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
|
|
311
|
+
additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
312
|
+
suggestResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
313
|
+
forceResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
314
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
315
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
316
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
334
317
|
};
|
|
318
|
+
/** @internal */
|
|
319
|
+
protected static optionsSchema: z.ZodObject<{
|
|
320
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
321
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
322
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
323
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
324
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
325
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
326
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
327
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
328
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
329
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
330
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
331
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
332
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
333
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
334
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
335
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
336
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
337
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
338
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
339
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
340
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
341
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
342
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
343
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
344
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
345
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
346
|
+
deferred: "deferred";
|
|
347
|
+
writeThrough: "writeThrough";
|
|
348
|
+
}>>;
|
|
349
|
+
}, z.core.$strict>]>>;
|
|
350
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
351
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
352
|
+
httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
|
|
353
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
354
|
+
configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
|
|
355
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
356
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
357
|
+
eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
|
|
358
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
359
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
360
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
361
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
362
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
363
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
364
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
365
|
+
id: z.ZodOptional<z.ZodString>;
|
|
366
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
367
|
+
ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
|
|
368
|
+
additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
369
|
+
suggestResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
370
|
+
forceResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
371
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
372
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
373
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
374
|
+
}, z.core.$strict>;
|
|
335
375
|
/**
|
|
336
376
|
* All `HttpCrawlerOptions` parameters are passed via an options object.
|
|
337
377
|
*/
|
|
338
|
-
constructor(options?: HttpCrawlerOptions<Context
|
|
339
|
-
/**
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
protected _runRequestHandler(crawlingContext: Context): Promise<void>;
|
|
349
|
-
protected isRequestBlocked(crawlingContext: Context): Promise<string | false>;
|
|
350
|
-
protected _handleNavigation(crawlingContext: Context): Promise<void>;
|
|
351
|
-
/**
|
|
352
|
-
* Sets the cookie header to `gotOptions` based on the provided request and session headers, as well as any changes that occurred due to hooks.
|
|
353
|
-
*/
|
|
354
|
-
protected _applyCookies({ session, request }: CrawlingContext, gotOptions: OptionsInit, preHookCookies: string, postHookCookies: string): void;
|
|
378
|
+
constructor(options?: HttpCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> & RequireContextPipeline<InternalHttpCrawlingContext, Context>);
|
|
379
|
+
/** @internal */
|
|
380
|
+
protected getNavigationTimeoutMillis(): number;
|
|
381
|
+
/** @internal */
|
|
382
|
+
protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem;
|
|
383
|
+
protected buildContextPipeline(): ContextPipeline<CrawlingContext, InternalHttpCrawlingContext>;
|
|
384
|
+
private prepareHttpRequest;
|
|
385
|
+
private makeHttpRequest;
|
|
386
|
+
private processHttpResponse;
|
|
387
|
+
private handleBlockedRequestByContent;
|
|
355
388
|
/**
|
|
356
389
|
* Function to make the HTTP request. It performs optimizations
|
|
357
390
|
* on the request such as only downloading the request body if the
|
|
358
391
|
* received content type matches text/html, application/xml, application/xhtml+xml.
|
|
359
392
|
*/
|
|
360
|
-
|
|
393
|
+
private requestFunction;
|
|
361
394
|
/**
|
|
362
395
|
* Encodes and parses response according to the provided content type
|
|
363
396
|
*/
|
|
364
|
-
|
|
365
|
-
isXml: boolean;
|
|
366
|
-
response: IncomingMessage;
|
|
367
|
-
contentType: {
|
|
368
|
-
type: string;
|
|
369
|
-
encoding: BufferEncoding;
|
|
370
|
-
};
|
|
371
|
-
}) | {
|
|
372
|
-
body: Buffer<ArrayBufferLike>;
|
|
373
|
-
response: IncomingMessage;
|
|
374
|
-
contentType: {
|
|
375
|
-
type: string;
|
|
376
|
-
encoding: BufferEncoding;
|
|
377
|
-
};
|
|
378
|
-
enqueueLinks: () => Promise<{
|
|
379
|
-
processedRequests: never[];
|
|
380
|
-
unprocessedRequests: never[];
|
|
381
|
-
}>;
|
|
382
|
-
}>;
|
|
383
|
-
protected _parseHTML(response: IncomingMessage, _isXml: boolean, _crawlingContext: Context): Promise<Partial<Context>>;
|
|
397
|
+
private parseResponse;
|
|
384
398
|
/**
|
|
385
399
|
* Combines the provided `requestOptions` with mandatory (non-overridable) values.
|
|
386
400
|
*/
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
body?: string | Buffer | Readable | Generator | AsyncGenerator | import("form-data-encoder").FormDataLike | undefined;
|
|
390
|
-
json?: unknown;
|
|
391
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
392
|
-
request?: import("got-scraping").RequestFunction | undefined;
|
|
393
|
-
url?: string | URL | undefined;
|
|
394
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
395
|
-
headers?: import("got-scraping").Headers | undefined;
|
|
396
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
397
|
-
agent?: import("got-scraping").Agents | undefined;
|
|
398
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
399
|
-
h2session?: import("http2").ClientHttp2Session | undefined;
|
|
400
|
-
decompress?: boolean | undefined;
|
|
401
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
402
|
-
timeout?: import("got-scraping").Delays | undefined;
|
|
403
|
-
prefixUrl?: string | URL | undefined;
|
|
404
|
-
form?: Record<string, any> | undefined;
|
|
405
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
406
|
-
cookieJar?: import("got-scraping").PromiseCookieJar | import("got-scraping").ToughCookieJar | undefined;
|
|
407
|
-
signal?: AbortSignal | undefined;
|
|
408
|
-
ignoreInvalidCookies?: boolean | undefined;
|
|
409
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
410
|
-
searchParams?: string | import("got-scraping").SearchParameters | URLSearchParams | undefined;
|
|
411
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
412
|
-
dnsLookup?: import("cacheable-lookup").default["lookup"] | undefined;
|
|
413
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
414
|
-
dnsCache?: import("cacheable-lookup").default | boolean | undefined;
|
|
415
|
-
context?: Record<string, unknown> | undefined;
|
|
416
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
417
|
-
followRedirect?: boolean | ((response: import("got-scraping").PlainResponse) => boolean) | undefined;
|
|
418
|
-
maxRedirects?: number | undefined;
|
|
419
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
420
|
-
cache?: string | import("cacheable-request").StorageAdapter | boolean | undefined;
|
|
421
|
-
throwHttpErrors?: boolean | undefined;
|
|
422
|
-
username?: string | undefined;
|
|
423
|
-
password?: string | undefined;
|
|
424
|
-
http2?: boolean | undefined;
|
|
425
|
-
allowGetBody?: boolean | undefined;
|
|
426
|
-
methodRewriting?: boolean | undefined;
|
|
427
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
428
|
-
dnsLookupIpVersion?: import("got-scraping").DnsLookupIpVersion;
|
|
429
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
430
|
-
parseJson?: import("got-scraping").ParseJsonFunction | undefined;
|
|
431
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
432
|
-
stringifyJson?: import("got-scraping").StringifyJsonFunction | undefined;
|
|
433
|
-
localAddress?: string | undefined;
|
|
434
|
-
method?: Method | undefined;
|
|
435
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
436
|
-
createConnection?: import("got-scraping").CreateConnectionFunction | undefined;
|
|
437
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
438
|
-
cacheOptions?: import("got-scraping").CacheOptions | undefined;
|
|
439
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
440
|
-
https?: import("got-scraping").HttpsOptions | undefined;
|
|
441
|
-
encoding?: BufferEncoding | undefined;
|
|
442
|
-
resolveBodyOnly?: boolean | undefined;
|
|
443
|
-
isStream?: boolean | undefined;
|
|
444
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
445
|
-
responseType?: import("got-scraping").ResponseType | undefined;
|
|
446
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
447
|
-
pagination?: import("got-scraping").PaginationOptions<unknown, unknown> | undefined;
|
|
448
|
-
setHost?: boolean | undefined;
|
|
449
|
-
maxHeaderSize?: number | undefined;
|
|
450
|
-
enableUnixSockets?: boolean | undefined;
|
|
451
|
-
} & {
|
|
452
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
453
|
-
hooks?: Partial<import("got-scraping").Hooks>;
|
|
454
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
455
|
-
retry?: Partial<import("got-scraping").RetryOptions>;
|
|
456
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
457
|
-
} & import("got-scraping").Context & Required<Pick<OptionsInit, "url">> & {
|
|
458
|
-
isStream: true;
|
|
459
|
-
};
|
|
460
|
-
protected _encodeResponse(request: Request, response: IncomingMessage, encoding: BufferEncoding): {
|
|
461
|
-
encoding: BufferEncoding;
|
|
462
|
-
response: IncomingMessage;
|
|
463
|
-
};
|
|
401
|
+
private getRequestOptions;
|
|
402
|
+
private encodeResponse;
|
|
464
403
|
/**
|
|
465
404
|
* Checks and extends supported mime types
|
|
466
405
|
*/
|
|
467
|
-
|
|
406
|
+
private extendSupportedMimeTypes;
|
|
468
407
|
/**
|
|
469
408
|
* Handles timeout request
|
|
470
409
|
*/
|
|
471
|
-
|
|
472
|
-
private
|
|
410
|
+
private handleRequestTimeout;
|
|
411
|
+
private abortDownloadOfBody;
|
|
473
412
|
/**
|
|
474
413
|
* @internal wraps public utility for mocking purposes
|
|
475
414
|
*/
|
|
476
|
-
private
|
|
477
|
-
}
|
|
478
|
-
interface RequestFunctionOptions {
|
|
479
|
-
request: Request;
|
|
480
|
-
session?: Session;
|
|
481
|
-
proxyUrl?: string;
|
|
482
|
-
gotOptions: OptionsInit;
|
|
415
|
+
private requestAsBrowser;
|
|
483
416
|
}
|
|
484
417
|
/**
|
|
485
418
|
* Creates new {@link Router} instance that works based on request labels.
|
|
@@ -505,7 +438,7 @@ interface RequestFunctionOptions {
|
|
|
505
438
|
* await crawler.run();
|
|
506
439
|
* ```
|
|
507
440
|
*/
|
|
508
|
-
|
|
509
|
-
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData
|
|
441
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
442
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
443
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
510
444
|
export {};
|
|
511
|
-
//# sourceMappingURL=http-crawler.d.ts.map
|