@crawlee/http 4.0.0-beta.18 → 4.0.0-beta.181
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +1 -1
- package/index.js +1 -1
- package/internals/dom-crawler.d.ts +131 -0
- package/internals/dom-crawler.js +101 -0
- package/internals/file-download.d.ts +14 -12
- package/internals/file-download.js +46 -40
- package/internals/http-crawler.d.ts +212 -263
- package/internals/http-crawler.js +253 -252
- package/internals/utils.d.ts +11 -1
- package/internals/utils.js +50 -6
- package/package.json +13 -12
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/file-download.d.ts.map +0 -1
- package/internals/file-download.js.map +0 -1
- package/internals/http-crawler.d.ts.map +0 -1
- package/internals/http-crawler.js.map +0 -1
- package/internals/utils.d.ts.map +0 -1
- package/internals/utils.js.map +0 -1
|
@@ -1,57 +1,86 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import
|
|
3
|
-
import { BasicCrawler, Configuration, ContextPipeline } from '@crawlee/basic';
|
|
4
|
-
import type { LoadedRequest } from '@crawlee/core';
|
|
1
|
+
import type { BasicCrawlerOptions, ConcurrencySystem, ConcurrencySystemOptions, CrawlingContext, ErrorHandler, GetUserDataFromRequest, LoadedRequest, Request as CrawleeRequest, RequestHandler, RequireContextPipeline, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/basic';
|
|
2
|
+
import { BasicCrawler, ContextPipeline } from '@crawlee/basic';
|
|
5
3
|
import type { Awaitable, Dictionary } from '@crawlee/types';
|
|
6
|
-
import {
|
|
7
|
-
import type { RequestLike, ResponseLike } from 'content-type';
|
|
8
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
9
|
-
import type { Method, OptionsInit } from 'got-scraping';
|
|
4
|
+
import type { CheerioAPI } from 'cheerio';
|
|
10
5
|
import type { JsonValue } from 'type-fest';
|
|
6
|
+
import { z } from 'zod';
|
|
7
|
+
/**
|
|
8
|
+
* A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event
|
|
9
|
+
* loop. {@link HttpCrawler} folds these into the {@link ConcurrencySystem} it builds by default.
|
|
10
|
+
*
|
|
11
|
+
* A {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default
|
|
12
|
+
* wholesale, tuning included, so spread these options in if you want to keep it:
|
|
13
|
+
*
|
|
14
|
+
* ```typescript
|
|
15
|
+
* new ConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, maxConcurrency: 50 });
|
|
16
|
+
* ```
|
|
17
|
+
*/
|
|
18
|
+
export declare const HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS: ConcurrencySystemOptions;
|
|
11
19
|
export type HttpErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
12
|
-
JSONData extends JsonValue = any
|
|
13
|
-
|
|
20
|
+
JSONData extends JsonValue = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
21
|
+
ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, HttpCrawlingContext<UserData, JSONData> & ContextExtension>;
|
|
22
|
+
export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
14
23
|
/**
|
|
15
|
-
* Timeout
|
|
24
|
+
* Timeout for the whole navigation phase, given in seconds. A single window shared by the
|
|
25
|
+
* `preNavigationHooks`, the navigation (the HTTP request to the resource), and the `postNavigationHooks` -
|
|
26
|
+
* so a slow hook eats into the same budget the navigation uses. Separate from the
|
|
27
|
+
* {@link BasicCrawlerOptions.requestHandlerTimeoutSecs|`requestHandlerTimeoutSecs`}, which times only the
|
|
28
|
+
* request handler.
|
|
16
29
|
*/
|
|
17
30
|
navigationTimeoutSecs?: number;
|
|
18
31
|
/**
|
|
19
|
-
* If set to true
|
|
32
|
+
* If set to `true`, TLS/SSL certificate errors are ignored. Forwarded to the HTTP client as
|
|
33
|
+
* {@link SendRequestOptions.ignoreTlsErrors|`ignoreTlsErrors`} on every navigation request, so custom
|
|
34
|
+
* {@link BaseHttpClient} implementations should honor that flag (the built-in impit and got-scraping
|
|
35
|
+
* clients do; the native fetch fallback cannot disable TLS verification and warns instead).
|
|
36
|
+
*
|
|
37
|
+
* @default true
|
|
20
38
|
*/
|
|
21
|
-
|
|
39
|
+
ignoreTlsErrors?: boolean;
|
|
22
40
|
/**
|
|
23
41
|
* Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
|
|
24
|
-
* or browser properties before navigation. The function accepts
|
|
25
|
-
* which
|
|
42
|
+
* or browser properties before navigation. The function accepts one parameter `crawlingContext`,
|
|
43
|
+
* which is passed to the `requestAsBrowser()` function the crawler calls to navigate.
|
|
44
|
+
*
|
|
45
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context,
|
|
46
|
+
* allowing the hook to override context members for subsequent hooks and pipeline stages.
|
|
47
|
+
*
|
|
48
|
+
* The context is built up in the following order: base context (`request`, `session`, helpers, ...) ->
|
|
49
|
+
* `extendContext` -> `preNavigationHooks` -> navigation -> `postNavigationHooks` -> `requestHandler`.
|
|
50
|
+
* This means the members added by `extendContext` are already available here, but navigation-dependent
|
|
51
|
+
* members (e.g. `response`, `body`, `$`) are not.
|
|
26
52
|
* Example:
|
|
27
53
|
* ```
|
|
28
54
|
* preNavigationHooks: [
|
|
29
|
-
* async (crawlingContext
|
|
55
|
+
* async (crawlingContext) => {
|
|
30
56
|
* // ...
|
|
31
57
|
* },
|
|
32
58
|
* ]
|
|
33
59
|
* ```
|
|
34
|
-
*
|
|
35
|
-
* Modyfing `pageOptions` is supported only in Playwright incognito.
|
|
36
|
-
* See {@link PrePageCreateHook}
|
|
37
60
|
*/
|
|
38
|
-
preNavigationHooks?: InternalHttpHook<CrawlingContext>[];
|
|
61
|
+
preNavigationHooks?: InternalHttpHook<CrawlingContext<any>, ContextExtension>[];
|
|
39
62
|
/**
|
|
40
63
|
* Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
|
|
41
64
|
* The function accepts `crawlingContext` as the only parameter.
|
|
65
|
+
*
|
|
66
|
+
* A hook may optionally return a partial object whose properties are merged into the crawling context,
|
|
67
|
+
* which is useful for overriding the `response` after solving a challenge or re-fetching the resource.
|
|
42
68
|
* Example:
|
|
43
69
|
* ```
|
|
44
70
|
* postNavigationHooks: [
|
|
45
71
|
* async (crawlingContext) => {
|
|
46
|
-
*
|
|
72
|
+
* if (await needsRevalidation(crawlingContext)) {
|
|
73
|
+
* return { response: await refetch(crawlingContext.request) };
|
|
74
|
+
* }
|
|
47
75
|
* },
|
|
48
76
|
* ]
|
|
49
77
|
* ```
|
|
50
78
|
*/
|
|
51
|
-
postNavigationHooks?: ((crawlingContext:
|
|
79
|
+
postNavigationHooks?: ((crawlingContext: CrawlingContextWithResponse & ContextExtension) => Awaitable<void | Partial<CrawlingContextWithResponse>>)[];
|
|
52
80
|
/**
|
|
53
81
|
* An array of [MIME types](https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Complete_list_of_MIME_types)
|
|
54
|
-
* you want the crawler to load and process. By default, only `text/html
|
|
82
|
+
* you want the crawler to load and process. By default, only `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
83
|
+
* and `application/json` MIME types are supported.
|
|
55
84
|
*/
|
|
56
85
|
additionalMimeTypes?: string[];
|
|
57
86
|
/**
|
|
@@ -77,30 +106,17 @@ export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext
|
|
|
77
106
|
*/
|
|
78
107
|
forceResponseEncoding?: string;
|
|
79
108
|
/**
|
|
80
|
-
* Automatically saves cookies to Session.
|
|
109
|
+
* Automatically saves cookies to Session. Enabled by default.
|
|
81
110
|
*
|
|
82
111
|
* It parses cookie from response "set-cookie" header saves or updates cookies for session and once the session is used for next request.
|
|
83
112
|
* It passes the "Cookie" header to the request with the session cookies.
|
|
84
113
|
*/
|
|
85
|
-
|
|
86
|
-
/**
|
|
87
|
-
* An array of HTTP response [Status Codes](https://developer.mozilla.org/en-US/docs/Web/HTTP/Status) to be excluded from error consideration.
|
|
88
|
-
* By default, status codes >= 500 trigger errors.
|
|
89
|
-
*/
|
|
90
|
-
ignoreHttpErrorStatusCodes?: number[];
|
|
91
|
-
/**
|
|
92
|
-
* An array of additional HTTP response [Status Codes](https://developer.mozilla.org/en-US/docs/Web/HTTP/Status) to be treated as errors.
|
|
93
|
-
* By default, status codes >= 500 trigger errors.
|
|
94
|
-
*/
|
|
95
|
-
additionalHttpErrorStatusCodes?: number[];
|
|
114
|
+
saveResponseCookies?: boolean;
|
|
96
115
|
}
|
|
97
|
-
|
|
98
|
-
* @internal
|
|
99
|
-
*/
|
|
100
|
-
export type InternalHttpHook<Context> = (crawlingContext: Context, gotOptions: OptionsInit) => Awaitable<void>;
|
|
116
|
+
export type InternalHttpHook<Context, ContextExtension = {}> = (crawlingContext: Context & ContextExtension) => Awaitable<void | Partial<Context>>;
|
|
101
117
|
export type HttpHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
102
118
|
JSONData extends JsonValue = any> = InternalHttpHook<HttpCrawlingContext<UserData, JSONData>>;
|
|
103
|
-
interface
|
|
119
|
+
interface CrawlingContextWithResponse<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
|
|
104
120
|
/**
|
|
105
121
|
* The request object that was successfully loaded and navigated to, including the {@link Request.loadedUrl|`loadedUrl`} property.
|
|
106
122
|
*/
|
|
@@ -110,11 +126,8 @@ interface CrawlingContextWithReponse<UserData extends Dictionary = any> extends
|
|
|
110
126
|
*/
|
|
111
127
|
response: Response;
|
|
112
128
|
}
|
|
113
|
-
/**
|
|
114
|
-
* @internal
|
|
115
|
-
*/
|
|
116
129
|
export interface InternalHttpCrawlingContext<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
117
|
-
JSONData extends JsonValue = any> extends
|
|
130
|
+
JSONData extends JsonValue = any> extends CrawlingContextWithResponse<UserData> {
|
|
118
131
|
/**
|
|
119
132
|
* The request body of the web page.
|
|
120
133
|
* The type depends on the `Content-Type` header of the web page:
|
|
@@ -158,7 +171,7 @@ JSONData extends JsonValue = any> extends CrawlingContextWithReponse<UserData> {
|
|
|
158
171
|
* });
|
|
159
172
|
* ```
|
|
160
173
|
*/
|
|
161
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<
|
|
174
|
+
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
162
175
|
}
|
|
163
176
|
export interface HttpCrawlingContext<UserData extends Dictionary = any, JSONData extends JsonValue = any> extends InternalHttpCrawlingContext<UserData, JSONData> {
|
|
164
177
|
}
|
|
@@ -175,38 +188,40 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
|
|
|
175
188
|
*
|
|
176
189
|
* This crawler downloads each URL using a plain HTTP request and doesn't do any HTML parsing.
|
|
177
190
|
*
|
|
178
|
-
* The source URLs are represented using {@link Request} objects that are fed from
|
|
179
|
-
* {@link
|
|
180
|
-
*
|
|
191
|
+
* The source URLs are represented using {@link Request} objects that are fed from the
|
|
192
|
+
* {@link IRequestManager|request manager} provided via the {@link HttpCrawlerOptions.requestManager|`requestManager`}
|
|
193
|
+
* constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
|
|
194
|
+
* as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
|
|
195
|
+
* {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
|
|
196
|
+
* result as `requestManager`.
|
|
181
197
|
*
|
|
182
|
-
*
|
|
183
|
-
*
|
|
184
|
-
* to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
|
|
198
|
+
* > The {@link HttpCrawlerOptions.requestList|`requestList`} and {@link HttpCrawlerOptions.requestQueue|`requestQueue`}
|
|
199
|
+
* > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
|
|
185
200
|
*
|
|
186
201
|
* The crawler finishes when there are no more {@link Request} objects to crawl.
|
|
187
202
|
*
|
|
188
|
-
* We can use the `preNavigationHooks` to adjust
|
|
203
|
+
* We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
|
|
189
204
|
*
|
|
190
205
|
* ```javascript
|
|
191
206
|
* preNavigationHooks: [
|
|
192
|
-
* (crawlingContext
|
|
207
|
+
* (crawlingContext) => {
|
|
193
208
|
* // ...
|
|
194
209
|
* },
|
|
195
210
|
* ]
|
|
196
211
|
* ```
|
|
197
212
|
*
|
|
198
|
-
* By default, this crawler only processes web pages with the `text/html`
|
|
199
|
-
* and `application/
|
|
213
|
+
* By default, this crawler only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
|
|
214
|
+
* and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
|
|
200
215
|
* and skips pages with other content types. If you want the crawler to process other content types,
|
|
201
216
|
* use the {@link HttpCrawlerOptions.additionalMimeTypes} constructor option.
|
|
202
217
|
* Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
|
|
203
218
|
* For details, see {@link HttpCrawlerOptions.requestHandler}.
|
|
204
219
|
*
|
|
205
|
-
* New requests are only dispatched when there is enough free CPU and memory available,
|
|
206
|
-
*
|
|
207
|
-
*
|
|
208
|
-
*
|
|
209
|
-
* {@link
|
|
220
|
+
* New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
|
|
221
|
+
* {@link ConcurrencySystem}.
|
|
222
|
+
* Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
|
|
223
|
+
* constructor, or, for finer control, by injecting a pre-configured
|
|
224
|
+
* {@link ConcurrencySystem|`concurrencySystem`}.
|
|
210
225
|
*
|
|
211
226
|
* **Example usage:**
|
|
212
227
|
*
|
|
@@ -231,236 +246,170 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
|
|
|
231
246
|
* ```
|
|
232
247
|
* @category Crawlers
|
|
233
248
|
*/
|
|
234
|
-
export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any> = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension> extends BasicCrawler<Context, ContextExtension, ExtendedContext> {
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
protected navigationTimeoutMillis: number;
|
|
240
|
-
protected ignoreSslErrors: boolean;
|
|
241
|
-
protected suggestResponseEncoding?: string;
|
|
242
|
-
protected forceResponseEncoding?: string;
|
|
243
|
-
protected additionalHttpErrorStatusCodes: Set<number>;
|
|
244
|
-
protected ignoreHttpErrorStatusCodes: Set<number>;
|
|
245
|
-
protected readonly supportedMimeTypes: Set<string>;
|
|
249
|
+
export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any> = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawler<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
250
|
+
#private;
|
|
251
|
+
/**
|
|
252
|
+
* @internal
|
|
253
|
+
*/
|
|
246
254
|
protected static optionsShape: {
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
302
|
-
statusMessageLoggingInterval: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
303
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
304
|
-
statusMessageCallback: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
|
|
305
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
306
|
-
retryOnBlocked: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
307
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
308
|
-
respectRobotsTxtFile: import("ow").AnyPredicate<boolean | object>;
|
|
309
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
310
|
-
onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
|
|
311
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
312
|
-
httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
313
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
314
|
-
minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
315
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
316
|
-
maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
317
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
318
|
-
maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
|
|
319
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
320
|
-
keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
|
|
321
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
322
|
-
log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
323
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
324
|
-
experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
325
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
326
|
-
statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
|
|
255
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
256
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
257
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
258
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
259
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
260
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
261
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
262
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
263
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
264
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
265
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
266
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
267
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
268
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
269
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
270
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
271
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
272
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
273
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
274
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
275
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
276
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
277
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
278
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
279
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
280
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
281
|
+
deferred: "deferred";
|
|
282
|
+
writeThrough: "writeThrough";
|
|
283
|
+
}>>;
|
|
284
|
+
}, z.core.$strict>]>>;
|
|
285
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
286
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
287
|
+
httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
|
|
288
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
289
|
+
configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
|
|
290
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
291
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
292
|
+
eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
|
|
293
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
294
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
295
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
296
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
297
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
298
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
299
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
300
|
+
id: z.ZodOptional<z.ZodString>;
|
|
301
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
302
|
+
ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
|
|
303
|
+
additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
304
|
+
suggestResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
305
|
+
forceResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
306
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
307
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
308
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
327
309
|
};
|
|
310
|
+
/** @internal */
|
|
311
|
+
protected static optionsSchema: z.ZodObject<{
|
|
312
|
+
contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
313
|
+
extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
314
|
+
requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
315
|
+
requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
316
|
+
requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
317
|
+
requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
318
|
+
requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
319
|
+
errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
320
|
+
failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
321
|
+
maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
322
|
+
sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
323
|
+
maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
324
|
+
maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
325
|
+
taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
326
|
+
concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
327
|
+
sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
328
|
+
proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
329
|
+
statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
330
|
+
statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
331
|
+
additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
332
|
+
ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
333
|
+
blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
|
|
334
|
+
retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
|
|
335
|
+
respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
|
|
336
|
+
transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
|
|
337
|
+
requestQueue: z.ZodOptional<z.ZodEnum<{
|
|
338
|
+
deferred: "deferred";
|
|
339
|
+
writeThrough: "writeThrough";
|
|
340
|
+
}>>;
|
|
341
|
+
}, z.core.$strict>]>>;
|
|
342
|
+
onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
|
|
343
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
344
|
+
httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
|
|
345
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
346
|
+
configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
|
|
347
|
+
storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
348
|
+
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
349
|
+
eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
|
|
350
|
+
logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
|
|
351
|
+
minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
352
|
+
maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
353
|
+
initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
354
|
+
maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
|
|
355
|
+
keepAlive: z.ZodOptional<z.ZodBoolean>;
|
|
356
|
+
statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
|
|
357
|
+
id: z.ZodOptional<z.ZodString>;
|
|
358
|
+
navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
|
|
359
|
+
ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
|
|
360
|
+
additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
|
|
361
|
+
suggestResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
362
|
+
forceResponseEncoding: z.ZodOptional<z.ZodString>;
|
|
363
|
+
saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
|
|
364
|
+
preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
365
|
+
postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
|
|
366
|
+
}, z.core.$strict>;
|
|
328
367
|
/**
|
|
329
368
|
* All `HttpCrawlerOptions` parameters are passed via an options object.
|
|
330
369
|
*/
|
|
331
|
-
constructor(options?: HttpCrawlerOptions<Context, ContextExtension, ExtendedContext> & RequireContextPipeline<InternalHttpCrawlingContext, Context
|
|
370
|
+
constructor(options?: HttpCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> & RequireContextPipeline<InternalHttpCrawlingContext, Context>);
|
|
371
|
+
protected getNavigationTimeoutMillis(): number;
|
|
372
|
+
/**
|
|
373
|
+
* Folds {@link HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS} into the default system, keeping the user's
|
|
374
|
+
* concurrency shortcuts on top. Not called for a supplied
|
|
375
|
+
* {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} — spread the constant into it yourself to
|
|
376
|
+
* keep the tuning.
|
|
377
|
+
*/
|
|
378
|
+
protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem;
|
|
332
379
|
protected buildContextPipeline(): ContextPipeline<CrawlingContext, InternalHttpCrawlingContext>;
|
|
380
|
+
private prepareHttpRequest;
|
|
333
381
|
private makeHttpRequest;
|
|
334
382
|
private processHttpResponse;
|
|
335
383
|
private handleBlockedRequestByContent;
|
|
336
384
|
protected isRequestBlocked(crawlingContext: InternalHttpCrawlingContext): Promise<string | false>;
|
|
337
|
-
/**
|
|
338
|
-
* Sets the cookie header to `gotOptions` based on the provided request and session headers, as well as any changes that occurred due to hooks.
|
|
339
|
-
*/
|
|
340
|
-
protected _applyCookies({ session, request }: CrawlingContext, gotOptions: OptionsInit, preHookCookies: string, postHookCookies: string): void;
|
|
341
385
|
/**
|
|
342
386
|
* Function to make the HTTP request. It performs optimizations
|
|
343
387
|
* on the request such as only downloading the request body if the
|
|
344
388
|
* received content type matches text/html, application/xml, application/xhtml+xml.
|
|
345
389
|
*/
|
|
346
|
-
|
|
390
|
+
private requestFunction;
|
|
347
391
|
/**
|
|
348
392
|
* Encodes and parses response according to the provided content type
|
|
349
393
|
*/
|
|
350
|
-
|
|
351
|
-
response: Response;
|
|
352
|
-
contentType: {
|
|
353
|
-
type: string;
|
|
354
|
-
encoding: BufferEncoding;
|
|
355
|
-
};
|
|
356
|
-
body: string;
|
|
357
|
-
} | {
|
|
358
|
-
body: Buffer<ArrayBuffer>;
|
|
359
|
-
response: Response;
|
|
360
|
-
contentType: {
|
|
361
|
-
type: string;
|
|
362
|
-
encoding: BufferEncoding;
|
|
363
|
-
};
|
|
364
|
-
}>;
|
|
394
|
+
private parseResponse;
|
|
365
395
|
/**
|
|
366
396
|
* Combines the provided `requestOptions` with mandatory (non-overridable) values.
|
|
367
397
|
*/
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
371
|
-
headers?: import("got-scraping").Headers | undefined;
|
|
372
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
373
|
-
request?: import("got-scraping").RequestFunction | undefined;
|
|
374
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
375
|
-
agent?: import("got-scraping").Agents | undefined;
|
|
376
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
377
|
-
h2session?: import("http2").ClientHttp2Session | undefined;
|
|
378
|
-
decompress?: boolean | undefined;
|
|
379
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
380
|
-
timeout?: import("got-scraping").Delays | undefined;
|
|
381
|
-
prefixUrl?: string | URL | undefined;
|
|
382
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
383
|
-
body?: string | Buffer | Readable | Generator | AsyncGenerator | import("form-data-encoder").FormDataLike | undefined;
|
|
384
|
-
form?: Record<string, any> | undefined;
|
|
385
|
-
json?: unknown;
|
|
386
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
387
|
-
cookieJar?: import("got-scraping").PromiseCookieJar | import("got-scraping").ToughCookieJar | undefined;
|
|
388
|
-
signal?: AbortSignal | undefined;
|
|
389
|
-
ignoreInvalidCookies?: boolean | undefined;
|
|
390
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
391
|
-
searchParams?: string | import("got-scraping").SearchParameters | URLSearchParams | undefined;
|
|
392
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
393
|
-
dnsLookup?: import("cacheable-lookup").default["lookup"] | undefined;
|
|
394
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
395
|
-
dnsCache?: import("cacheable-lookup").default | boolean | undefined;
|
|
396
|
-
context?: Record<string, unknown> | undefined;
|
|
397
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
398
|
-
followRedirect?: boolean | ((response: import("got-scraping").PlainResponse) => boolean) | undefined;
|
|
399
|
-
maxRedirects?: number | undefined;
|
|
400
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
401
|
-
cache?: string | import("cacheable-request").StorageAdapter | boolean | undefined;
|
|
402
|
-
throwHttpErrors?: boolean | undefined;
|
|
403
|
-
username?: string | undefined;
|
|
404
|
-
password?: string | undefined;
|
|
405
|
-
http2?: boolean | undefined;
|
|
406
|
-
allowGetBody?: boolean | undefined;
|
|
407
|
-
methodRewriting?: boolean | undefined;
|
|
408
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
409
|
-
dnsLookupIpVersion?: import("got-scraping").DnsLookupIpVersion;
|
|
410
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
411
|
-
parseJson?: import("got-scraping").ParseJsonFunction | undefined;
|
|
412
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
413
|
-
stringifyJson?: import("got-scraping").StringifyJsonFunction | undefined;
|
|
414
|
-
localAddress?: string | undefined;
|
|
415
|
-
method?: Method | undefined;
|
|
416
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
417
|
-
createConnection?: import("got-scraping").CreateConnectionFunction | undefined;
|
|
418
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
419
|
-
cacheOptions?: import("got-scraping").CacheOptions | undefined;
|
|
420
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
421
|
-
https?: import("got-scraping").HttpsOptions | undefined;
|
|
422
|
-
encoding?: BufferEncoding | undefined;
|
|
423
|
-
resolveBodyOnly?: boolean | undefined;
|
|
424
|
-
isStream?: boolean | undefined;
|
|
425
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
426
|
-
responseType?: import("got-scraping").ResponseType | undefined;
|
|
427
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
428
|
-
pagination?: import("got-scraping").PaginationOptions<unknown, unknown> | undefined;
|
|
429
|
-
setHost?: boolean | undefined;
|
|
430
|
-
maxHeaderSize?: number | undefined;
|
|
431
|
-
enableUnixSockets?: boolean | undefined;
|
|
432
|
-
} & {
|
|
433
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
434
|
-
hooks?: Partial<import("got-scraping").Hooks>;
|
|
435
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
436
|
-
retry?: Partial<import("got-scraping").RetryOptions>;
|
|
437
|
-
// @ts-ignore optional peer dependency or compatibility with es2022
|
|
438
|
-
} & import("got-scraping").Context & Required<Pick<OptionsInit, "url">> & {
|
|
439
|
-
isStream: true;
|
|
440
|
-
};
|
|
441
|
-
protected _encodeResponse(request: CrawleeRequest, response: Response, encoding: BufferEncoding): {
|
|
442
|
-
encoding: BufferEncoding;
|
|
443
|
-
response: Response;
|
|
444
|
-
};
|
|
398
|
+
private getRequestOptions;
|
|
399
|
+
private encodeResponse;
|
|
445
400
|
/**
|
|
446
401
|
* Checks and extends supported mime types
|
|
447
402
|
*/
|
|
448
|
-
|
|
403
|
+
private extendSupportedMimeTypes;
|
|
449
404
|
/**
|
|
450
405
|
* Handles timeout request
|
|
451
406
|
*/
|
|
452
|
-
|
|
453
|
-
private
|
|
407
|
+
private handleRequestTimeout;
|
|
408
|
+
private abortDownloadOfBody;
|
|
454
409
|
/**
|
|
455
410
|
* @internal wraps public utility for mocking purposes
|
|
456
411
|
*/
|
|
457
|
-
private
|
|
458
|
-
}
|
|
459
|
-
interface RequestFunctionOptions {
|
|
460
|
-
request: CrawleeRequest;
|
|
461
|
-
session?: Session;
|
|
462
|
-
proxyUrl?: string;
|
|
463
|
-
gotOptions: OptionsInit;
|
|
412
|
+
private requestAsBrowser;
|
|
464
413
|
}
|
|
465
414
|
/**
|
|
466
415
|
* Creates new {@link Router} instance that works based on request labels.
|
|
@@ -486,7 +435,7 @@ interface RequestFunctionOptions {
|
|
|
486
435
|
* await crawler.run();
|
|
487
436
|
* ```
|
|
488
437
|
*/
|
|
489
|
-
|
|
490
|
-
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData
|
|
438
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
|
|
439
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
|
|
440
|
+
export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
|
|
491
441
|
export {};
|
|
492
|
-
//# sourceMappingURL=http-crawler.d.ts.map
|