@crawlee/http 4.0.0-beta.19 → 4.0.0-beta.191

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,25 +1,54 @@
1
- import type { BasicCrawlerOptions, CrawlingContext, ErrorHandler, GetUserDataFromRequest, Request as CrawleeRequest, RequestHandler, RequireContextPipeline, RouterRoutes, Session } from '@crawlee/basic';
2
- import { BasicCrawler, Configuration, ContextPipeline } from '@crawlee/basic';
3
- import type { LoadedRequest } from '@crawlee/core';
1
+ import type { BasicCrawlerOptions, ConcurrencySystem, ConcurrencySystemOptions, CrawlingContext, ErrorHandler, GetUserDataFromRequest, LoadedRequest, CrawlingRequest, RequestHandler, RequireContextPipeline, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/basic';
2
+ import { BasicCrawler, ContextPipeline } from '@crawlee/basic';
4
3
  import type { Awaitable, Dictionary } from '@crawlee/types';
5
- import { type CheerioRoot } from '@crawlee/utils';
6
- import type { RequestLike, ResponseLike } from 'content-type';
4
+ import type { CheerioAPI } from 'cheerio';
7
5
  import type { JsonValue } from 'type-fest';
6
+ import { z } from 'zod';
7
+ /**
8
+ * A higher starting concurrency and a relaxed event loop signal, since HTTP-only crawling barely touches the event
9
+ * loop. {@link HttpCrawler} folds these into the {@link ConcurrencySystem} it builds by default.
10
+ *
11
+ * A {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} you supply yourself replaces that default
12
+ * wholesale, tuning included, so spread these options in if you want to keep it:
13
+ *
14
+ * ```typescript
15
+ * new ConcurrencySystem({ ...HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS, maxConcurrency: 50 });
16
+ * ```
17
+ */
18
+ export declare const HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS: ConcurrencySystemOptions;
8
19
  export type HttpErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
9
- JSONData extends JsonValue = any> = ErrorHandler<HttpCrawlingContext<UserData, JSONData>>;
10
- export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension> extends BasicCrawlerOptions<Context, ContextExtension, ExtendedContext> {
20
+ JSONData extends JsonValue = any, // with default to Dictionary we cant use a typed router in untyped crawler
21
+ ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, HttpCrawlingContext<UserData, JSONData> & ContextExtension>;
22
+ export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
11
23
  /**
12
- * Timeout in which the HTTP request to the resource needs to finish, given in seconds.
24
+ * Timeout for the whole navigation phase, given in seconds. A single window shared by the
25
+ * `preNavigationHooks`, the navigation (the HTTP request to the resource), and the `postNavigationHooks` -
26
+ * so a slow hook eats into the same budget the navigation uses. Separate from the
27
+ * {@link BasicCrawlerOptions.requestHandlerTimeoutSecs|`requestHandlerTimeoutSecs`}, which times only the
28
+ * request handler.
13
29
  */
14
30
  navigationTimeoutSecs?: number;
15
31
  /**
16
- * If set to true, SSL certificate errors will be ignored.
32
+ * If set to `true`, TLS/SSL certificate errors are ignored. Forwarded to the HTTP client as
33
+ * {@link SendRequestOptions.ignoreTlsErrors|`ignoreTlsErrors`} on every navigation request, so custom
34
+ * {@link BaseHttpClient} implementations should honor that flag (the built-in impit and got-scraping
35
+ * clients do; the native fetch fallback cannot disable TLS verification and warns instead).
36
+ *
37
+ * @default true
17
38
  */
18
- ignoreSslErrors?: boolean;
39
+ ignoreTlsErrors?: boolean;
19
40
  /**
20
41
  * Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
21
42
  * or browser properties before navigation. The function accepts one parameter `crawlingContext`,
22
43
  * which is passed to the `requestAsBrowser()` function the crawler calls to navigate.
44
+ *
45
+ * A hook may optionally return a partial object whose properties are merged into the crawling context,
46
+ * allowing the hook to override context members for subsequent hooks and pipeline stages.
47
+ *
48
+ * The context is built up in the following order: base context (`request`, `session`, helpers, ...) ->
49
+ * `extendContext` -> `preNavigationHooks` -> navigation -> `postNavigationHooks` -> `requestHandler`.
50
+ * This means the members added by `extendContext` are already available here, but navigation-dependent
51
+ * members (e.g. `response`, `body`, `$`) are not.
23
52
  * Example:
24
53
  * ```
25
54
  * preNavigationHooks: [
@@ -28,27 +57,30 @@ export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext
28
57
  * },
29
58
  * ]
30
59
  * ```
31
- *
32
- * Modyfing `pageOptions` is supported only in Playwright incognito.
33
- * See {@link PrePageCreateHook}
34
60
  */
35
- preNavigationHooks?: InternalHttpHook<CrawlingContext>[];
61
+ preNavigationHooks?: InternalHttpHook<CrawlingContext<any>, ContextExtension>[];
36
62
  /**
37
63
  * Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
38
64
  * The function accepts `crawlingContext` as the only parameter.
65
+ *
66
+ * A hook may optionally return a partial object whose properties are merged into the crawling context,
67
+ * which is useful for overriding the `response` after solving a challenge or re-fetching the resource.
39
68
  * Example:
40
69
  * ```
41
70
  * postNavigationHooks: [
42
71
  * async (crawlingContext) => {
43
- * // ...
72
+ * if (await needsRevalidation(crawlingContext)) {
73
+ * return { response: await refetch(crawlingContext.request) };
74
+ * }
44
75
  * },
45
76
  * ]
46
77
  * ```
47
78
  */
48
- postNavigationHooks?: ((crawlingContext: CrawlingContextWithReponse) => Awaitable<void>)[];
79
+ postNavigationHooks?: ((crawlingContext: CrawlingContextWithResponse & ContextExtension) => Awaitable<void | Partial<CrawlingContextWithResponse>>)[];
49
80
  /**
50
81
  * An array of [MIME types](https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Complete_list_of_MIME_types)
51
- * you want the crawler to load and process. By default, only `text/html` and `application/xhtml+xml` MIME types are supported.
82
+ * you want the crawler to load and process. By default, only `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
83
+ * and `application/json` MIME types are supported.
52
84
  */
53
85
  additionalMimeTypes?: string[];
54
86
  /**
@@ -74,44 +106,28 @@ export interface HttpCrawlerOptions<Context extends InternalHttpCrawlingContext
74
106
  */
75
107
  forceResponseEncoding?: string;
76
108
  /**
77
- * Automatically saves cookies to Session. Works only if Session Pool is used.
109
+ * Automatically saves cookies to Session. Enabled by default.
78
110
  *
79
111
  * It parses cookie from response "set-cookie" header saves or updates cookies for session and once the session is used for next request.
80
112
  * It passes the "Cookie" header to the request with the session cookies.
81
113
  */
82
- persistCookiesPerSession?: boolean;
83
- /**
84
- * An array of HTTP response [Status Codes](https://developer.mozilla.org/en-US/docs/Web/HTTP/Status) to be excluded from error consideration.
85
- * By default, status codes >= 500 trigger errors.
86
- */
87
- ignoreHttpErrorStatusCodes?: number[];
88
- /**
89
- * An array of additional HTTP response [Status Codes](https://developer.mozilla.org/en-US/docs/Web/HTTP/Status) to be treated as errors.
90
- * By default, status codes >= 500 trigger errors.
91
- */
92
- additionalHttpErrorStatusCodes?: number[];
114
+ saveResponseCookies?: boolean;
93
115
  }
94
- /**
95
- * @internal
96
- */
97
- export type InternalHttpHook<Context> = (crawlingContext: Context) => Awaitable<void>;
116
+ export type InternalHttpHook<Context, ContextExtension = {}> = (crawlingContext: Context & ContextExtension) => Awaitable<void | Partial<Context>>;
98
117
  export type HttpHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
99
118
  JSONData extends JsonValue = any> = InternalHttpHook<HttpCrawlingContext<UserData, JSONData>>;
100
- interface CrawlingContextWithReponse<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
119
+ interface CrawlingContextWithResponse<UserData extends Dictionary = any> extends CrawlingContext<UserData> {
101
120
  /**
102
121
  * The request object that was successfully loaded and navigated to, including the {@link Request.loadedUrl|`loadedUrl`} property.
103
122
  */
104
- request: LoadedRequest<CrawleeRequest<UserData>>;
123
+ request: LoadedRequest<CrawlingRequest<UserData>>;
105
124
  /**
106
125
  * The HTTP response object containing status code, headers, and other response metadata.
107
126
  */
108
127
  response: Response;
109
128
  }
110
- /**
111
- * @internal
112
- */
113
129
  export interface InternalHttpCrawlingContext<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
114
- JSONData extends JsonValue = any> extends CrawlingContextWithReponse<UserData> {
130
+ JSONData extends JsonValue = any> extends CrawlingContextWithResponse<UserData> {
115
131
  /**
116
132
  * The request body of the web page.
117
133
  * The type depends on the `Content-Type` header of the web page:
@@ -155,7 +171,7 @@ JSONData extends JsonValue = any> extends CrawlingContextWithReponse<UserData> {
155
171
  * });
156
172
  * ```
157
173
  */
158
- parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioRoot>;
174
+ parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
159
175
  }
160
176
  export interface HttpCrawlingContext<UserData extends Dictionary = any, JSONData extends JsonValue = any> extends InternalHttpCrawlingContext<UserData, JSONData> {
161
177
  }
@@ -172,13 +188,15 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
172
188
  *
173
189
  * This crawler downloads each URL using a plain HTTP request and doesn't do any HTML parsing.
174
190
  *
175
- * The source URLs are represented using {@link Request} objects that are fed from
176
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link HttpCrawlerOptions.requestList}
177
- * or {@link HttpCrawlerOptions.requestQueue} constructor options, respectively.
191
+ * The source URLs are represented using {@link Request} objects that are fed from the
192
+ * {@link IRequestManager|request manager} provided via the {@link HttpCrawlerOptions.requestManager|`requestManager`}
193
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
194
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
195
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
196
+ * result as `requestManager`.
178
197
  *
179
- * If both {@link HttpCrawlerOptions.requestList} and {@link HttpCrawlerOptions.requestQueue} are used,
180
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
181
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
198
+ * > The {@link HttpCrawlerOptions.requestList|`requestList`} and {@link HttpCrawlerOptions.requestQueue|`requestQueue`}
199
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
182
200
  *
183
201
  * The crawler finishes when there are no more {@link Request} objects to crawl.
184
202
  *
@@ -192,18 +210,18 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
192
210
  * ]
193
211
  * ```
194
212
  *
195
- * By default, this crawler only processes web pages with the `text/html`
196
- * and `application/xhtml+xml` MIME content types (as reported by the `Content-Type` HTTP header),
213
+ * By default, this crawler only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
214
+ * and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
197
215
  * and skips pages with other content types. If you want the crawler to process other content types,
198
216
  * use the {@link HttpCrawlerOptions.additionalMimeTypes} constructor option.
199
217
  * Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
200
218
  * For details, see {@link HttpCrawlerOptions.requestHandler}.
201
219
  *
202
- * New requests are only dispatched when there is enough free CPU and memory available,
203
- * using the functionality provided by the {@link AutoscaledPool} class.
204
- * All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions`
205
- * parameter of the constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
206
- * {@link AutoscaledPool} options are available directly in the constructor.
220
+ * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
221
+ * {@link ConcurrencySystem}.
222
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
223
+ * constructor, or, for finer control, by injecting a pre-configured
224
+ * {@link ConcurrencySystem|`concurrencySystem`}.
207
225
  *
208
226
  * **Example usage:**
209
227
  *
@@ -228,181 +246,170 @@ JSONData extends JsonValue = any> = RequestHandler<HttpCrawlingContext<UserData,
228
246
  * ```
229
247
  * @category Crawlers
230
248
  */
231
- export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any> = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension> extends BasicCrawler<Context, ContextExtension, ExtendedContext> {
232
- readonly config: Configuration;
233
- protected preNavigationHooks: InternalHttpHook<CrawlingContext>[];
234
- protected postNavigationHooks: ((crawlingContext: CrawlingContextWithReponse) => Awaitable<void>)[];
235
- protected persistCookiesPerSession: boolean;
236
- protected navigationTimeoutMillis: number;
237
- protected ignoreSslErrors: boolean;
238
- protected suggestResponseEncoding?: string;
239
- protected forceResponseEncoding?: string;
240
- protected additionalHttpErrorStatusCodes: Set<number>;
241
- protected ignoreHttpErrorStatusCodes: Set<number>;
242
- protected readonly supportedMimeTypes: Set<string>;
249
+ export declare class HttpCrawler<Context extends InternalHttpCrawlingContext<any, any> = InternalHttpCrawlingContext, ContextExtension = Dictionary<never>, ExtendedContext extends Context = Context & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>, StatisticStateExtension extends object = {}> extends BasicCrawler<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
250
+ #private;
251
+ /**
252
+ * @internal
253
+ */
243
254
  protected static optionsShape: {
244
- // @ts-ignore optional peer dependency or compatibility with es2022
245
- navigationTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
246
- // @ts-ignore optional peer dependency or compatibility with es2022
247
- ignoreSslErrors: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
248
- // @ts-ignore optional peer dependency or compatibility with es2022
249
- additionalMimeTypes: import("ow").ArrayPredicate<string>;
250
- // @ts-ignore optional peer dependency or compatibility with es2022
251
- suggestResponseEncoding: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
252
- // @ts-ignore optional peer dependency or compatibility with es2022
253
- forceResponseEncoding: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
254
- // @ts-ignore optional peer dependency or compatibility with es2022
255
- persistCookiesPerSession: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
256
- // @ts-ignore optional peer dependency or compatibility with es2022
257
- additionalHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
258
- // @ts-ignore optional peer dependency or compatibility with es2022
259
- ignoreHttpErrorStatusCodes: import("ow").ArrayPredicate<number>;
260
- // @ts-ignore optional peer dependency or compatibility with es2022
261
- preNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
262
- // @ts-ignore optional peer dependency or compatibility with es2022
263
- postNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
264
- // @ts-ignore optional peer dependency or compatibility with es2022
265
- contextPipelineBuilder: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
266
- // @ts-ignore optional peer dependency or compatibility with es2022
267
- extendContext: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
268
- // @ts-ignore optional peer dependency or compatibility with es2022
269
- requestList: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
270
- // @ts-ignore optional peer dependency or compatibility with es2022
271
- requestQueue: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
272
- // @ts-ignore optional peer dependency or compatibility with es2022
273
- requestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
274
- // @ts-ignore optional peer dependency or compatibility with es2022
275
- requestHandlerTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
276
- // @ts-ignore optional peer dependency or compatibility with es2022
277
- errorHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
278
- // @ts-ignore optional peer dependency or compatibility with es2022
279
- failedRequestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
280
- // @ts-ignore optional peer dependency or compatibility with es2022
281
- maxRequestRetries: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
282
- // @ts-ignore optional peer dependency or compatibility with es2022
283
- sameDomainDelaySecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
284
- // @ts-ignore optional peer dependency or compatibility with es2022
285
- maxSessionRotations: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
286
- // @ts-ignore optional peer dependency or compatibility with es2022
287
- maxRequestsPerCrawl: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
288
- // @ts-ignore optional peer dependency or compatibility with es2022
289
- maxCrawlDepth: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
290
- // @ts-ignore optional peer dependency or compatibility with es2022
291
- autoscaledPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
292
- // @ts-ignore optional peer dependency or compatibility with es2022
293
- sessionPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
294
- // @ts-ignore optional peer dependency or compatibility with es2022
295
- useSessionPool: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
296
- // @ts-ignore optional peer dependency or compatibility with es2022
297
- proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
298
- // @ts-ignore optional peer dependency or compatibility with es2022
299
- statusMessageLoggingInterval: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
300
- // @ts-ignore optional peer dependency or compatibility with es2022
301
- statusMessageCallback: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
302
- // @ts-ignore optional peer dependency or compatibility with es2022
303
- retryOnBlocked: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
304
- // @ts-ignore optional peer dependency or compatibility with es2022
305
- respectRobotsTxtFile: import("ow").AnyPredicate<boolean | object>;
306
- // @ts-ignore optional peer dependency or compatibility with es2022
307
- onSkippedRequest: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
308
- // @ts-ignore optional peer dependency or compatibility with es2022
309
- httpClient: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
310
- // @ts-ignore optional peer dependency or compatibility with es2022
311
- minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
312
- // @ts-ignore optional peer dependency or compatibility with es2022
313
- maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
314
- // @ts-ignore optional peer dependency or compatibility with es2022
315
- maxRequestsPerMinute: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
316
- // @ts-ignore optional peer dependency or compatibility with es2022
317
- keepAlive: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
318
- // @ts-ignore optional peer dependency or compatibility with es2022
319
- log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
320
- // @ts-ignore optional peer dependency or compatibility with es2022
321
- experiments: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
322
- // @ts-ignore optional peer dependency or compatibility with es2022
323
- statisticsOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
255
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
256
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
257
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
258
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
259
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
260
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
261
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
262
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
263
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
264
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
265
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
266
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
267
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
268
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
269
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
270
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
271
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
272
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
273
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
274
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
275
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
276
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
277
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
278
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
279
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
280
+ requestQueue: z.ZodOptional<z.ZodEnum<{
281
+ deferred: "deferred";
282
+ writeThrough: "writeThrough";
283
+ }>>;
284
+ }, z.core.$strict>]>>;
285
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
286
+ // @ts-ignore optional peer dependency or compatibility with es2022
287
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
288
+ // @ts-ignore optional peer dependency or compatibility with es2022
289
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
290
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
291
+ // @ts-ignore optional peer dependency or compatibility with es2022
292
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
293
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
294
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
295
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
296
+ initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
297
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
298
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
299
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
300
+ id: z.ZodOptional<z.ZodString>;
301
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
302
+ ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
303
+ additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
304
+ suggestResponseEncoding: z.ZodOptional<z.ZodString>;
305
+ forceResponseEncoding: z.ZodOptional<z.ZodString>;
306
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
307
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
308
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
324
309
  };
310
+ /** @internal */
311
+ protected static optionsSchema: z.ZodObject<{
312
+ contextPipelineBuilder: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
313
+ extendContext: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
314
+ requestList: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
315
+ requestQueue: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
316
+ requestManager: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
317
+ requestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
318
+ requestHandlerTimeoutSecs: z.ZodOptional<z.ZodCustom<number, number>>;
319
+ errorHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
320
+ failedRequestHandler: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
321
+ maxRequestRetries: z.ZodDefault<z.ZodCustom<number, number>>;
322
+ sameDomainDelaySecs: z.ZodDefault<z.ZodCustom<number, number>>;
323
+ maxRequestsPerCrawl: z.ZodOptional<z.ZodCustom<number, number>>;
324
+ maxCrawlDepth: z.ZodOptional<z.ZodCustom<number, number>>;
325
+ taskLoopOptions: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
326
+ concurrencySystem: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
327
+ sessionPool: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
328
+ proxyConfiguration: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
329
+ statusMessageLoggingInterval: z.ZodDefault<z.ZodCustom<number, number>>;
330
+ statusMessageCallback: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
331
+ additionalHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
332
+ ignoreHttpErrorStatusCodes: z.ZodDefault<z.ZodArray<z.ZodCustom<number, number>>>;
333
+ blockedStatusCodes: z.ZodOptional<z.ZodArray<z.ZodCustom<number, number>>>;
334
+ retryOnBlocked: z.ZodDefault<z.ZodBoolean>;
335
+ respectRobotsTxtFile: z.ZodDefault<z.ZodUnion<readonly [z.ZodBoolean, z.ZodCustom<Dictionary, Dictionary>]>>;
336
+ transactionalStorage: z.ZodOptional<z.ZodUnion<readonly [z.ZodBoolean, z.ZodObject<{
337
+ requestQueue: z.ZodOptional<z.ZodEnum<{
338
+ deferred: "deferred";
339
+ writeThrough: "writeThrough";
340
+ }>>;
341
+ }, z.core.$strict>]>>;
342
+ onSkippedRequest: z.ZodOptional<z.ZodCustom<(...args: any[]) => unknown, (...args: any[]) => unknown>>;
343
+ // @ts-ignore optional peer dependency or compatibility with es2022
344
+ httpClient: z.ZodOptional<z.ZodCustom<import("@crawlee/http-client").BaseHttpClient, import("@crawlee/http-client").BaseHttpClient>>;
345
+ // @ts-ignore optional peer dependency or compatibility with es2022
346
+ configuration: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").Configuration, import("@crawlee/basic").Configuration>>;
347
+ storageBackend: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
348
+ // @ts-ignore optional peer dependency or compatibility with es2022
349
+ eventManager: z.ZodOptional<z.ZodCustom<import("@crawlee/basic").EventManager, import("@crawlee/basic").EventManager>>;
350
+ logger: z.ZodOptional<z.ZodType<Dictionary<any>, unknown, z.core.$ZodTypeInternals<Dictionary<any>, unknown>>>;
351
+ minConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
352
+ maxConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
353
+ initialConcurrency: z.ZodOptional<z.ZodCustom<number, number>>;
354
+ maxRequestsPerMinute: z.ZodOptional<z.ZodCustom<number, number>>;
355
+ keepAlive: z.ZodOptional<z.ZodBoolean>;
356
+ statistics: z.ZodOptional<z.ZodCustom<Dictionary, Dictionary>>;
357
+ id: z.ZodOptional<z.ZodString>;
358
+ navigationTimeoutSecs: z.ZodDefault<z.ZodCustom<number, number>>;
359
+ ignoreTlsErrors: z.ZodDefault<z.ZodBoolean>;
360
+ additionalMimeTypes: z.ZodDefault<z.ZodArray<z.ZodString>>;
361
+ suggestResponseEncoding: z.ZodOptional<z.ZodString>;
362
+ forceResponseEncoding: z.ZodOptional<z.ZodString>;
363
+ saveResponseCookies: z.ZodDefault<z.ZodBoolean>;
364
+ preNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
365
+ postNavigationHooks: z.ZodDefault<z.ZodCustom<unknown[], unknown[]>>;
366
+ }, z.core.$strict>;
325
367
  /**
326
368
  * All `HttpCrawlerOptions` parameters are passed via an options object.
327
369
  */
328
- constructor(options?: HttpCrawlerOptions<Context, ContextExtension, ExtendedContext> & RequireContextPipeline<InternalHttpCrawlingContext, Context>, config?: Configuration);
370
+ constructor(options?: HttpCrawlerOptions<Context, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> & RequireContextPipeline<InternalHttpCrawlingContext, Context>);
371
+ protected getNavigationTimeoutMillis(): number;
372
+ /**
373
+ * Folds {@link HTTP_OPTIMIZED_CONCURRENCY_SYSTEM_OPTIONS} into the default system, keeping the user's
374
+ * concurrency shortcuts on top. Not called for a supplied
375
+ * {@link BasicCrawlerOptions.concurrencySystem|`concurrencySystem`} — spread the constant into it yourself to
376
+ * keep the tuning.
377
+ */
378
+ protected createDefaultConcurrencySystem(options: ConcurrencySystemOptions): ConcurrencySystem;
329
379
  protected buildContextPipeline(): ContextPipeline<CrawlingContext, InternalHttpCrawlingContext>;
380
+ private prepareHttpRequest;
330
381
  private makeHttpRequest;
331
382
  private processHttpResponse;
332
383
  private handleBlockedRequestByContent;
333
384
  protected isRequestBlocked(crawlingContext: InternalHttpCrawlingContext): Promise<string | false>;
334
- /**
335
- * Returns the `Cookie` header value based on the current context and
336
- * any changes that occurred in the navigation hooks.
337
- */
338
- protected _applyCookies({ session, request }: CrawlingContext, preHookCookies: string, postHookCookies: string): string;
339
385
  /**
340
386
  * Function to make the HTTP request. It performs optimizations
341
387
  * on the request such as only downloading the request body if the
342
388
  * received content type matches text/html, application/xml, application/xhtml+xml.
343
389
  */
344
- protected _requestFunction({ request, session, proxyUrl, cookieString, }: RequestFunctionOptions): Promise<Response>;
390
+ private requestFunction;
345
391
  /**
346
392
  * Encodes and parses response according to the provided content type
347
393
  */
348
- protected _parseResponse(request: CrawleeRequest, response: Response): Promise<{
349
- response: Response;
350
- contentType: {
351
- type: string;
352
- encoding: BufferEncoding;
353
- };
354
- body: string;
355
- } | {
356
- body: Buffer<ArrayBuffer>;
357
- response: Response;
358
- contentType: {
359
- type: string;
360
- encoding: BufferEncoding;
361
- };
362
- }>;
394
+ private parseResponse;
363
395
  /**
364
396
  * Combines the provided `requestOptions` with mandatory (non-overridable) values.
365
397
  */
366
- protected _getRequestOptions(request: CrawleeRequest, session?: Session, proxyUrl?: string): {
367
- url: string;
368
- // @ts-ignore optional peer dependency or compatibility with es2022
369
- method: import("@crawlee/types").AllowedHttpMethods;
370
- proxyUrl: string | undefined;
371
- timeout: {
372
- request: number;
373
- };
374
- // @ts-ignore optional peer dependency or compatibility with es2022
375
- cookieJar: import("tough-cookie").CookieJar | undefined;
376
- sessionToken: Session | undefined;
377
- headers: Record<string, string> | undefined;
378
- https: {
379
- rejectUnauthorized: boolean;
380
- };
381
- body: string | undefined;
382
- };
383
- protected _encodeResponse(request: CrawleeRequest, response: Response, encoding: BufferEncoding): {
384
- encoding: BufferEncoding;
385
- response: Response;
386
- };
398
+ private getRequestOptions;
399
+ private encodeResponse;
387
400
  /**
388
401
  * Checks and extends supported mime types
389
402
  */
390
- protected _extendSupportedMimeTypes(additionalMimeTypes: (string | RequestLike | ResponseLike)[]): void;
403
+ private extendSupportedMimeTypes;
391
404
  /**
392
405
  * Handles timeout request
393
406
  */
394
- protected _handleRequestTimeout(session?: Session): void;
395
- private _abortDownloadOfBody;
407
+ private handleRequestTimeout;
408
+ private abortDownloadOfBody;
396
409
  /**
397
410
  * @internal wraps public utility for mocking purposes
398
411
  */
399
- private _requestAsBrowser;
400
- }
401
- interface RequestFunctionOptions {
402
- request: CrawleeRequest;
403
- session?: Session;
404
- proxyUrl?: string;
405
- cookieString?: string;
412
+ private requestAsBrowser;
406
413
  }
407
414
  /**
408
415
  * Creates new {@link Router} instance that works based on request labels.
@@ -428,7 +435,7 @@ interface RequestFunctionOptions {
428
435
  * await crawler.run();
429
436
  * ```
430
437
  */
431
- // @ts-ignore optional peer dependency or compatibility with es2022
432
- export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData>): import("@crawlee/basic").RouterHandler<Context>;
438
+ export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
439
+ export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
440
+ export declare function createHttpRouter<Context extends HttpCrawlingContext = HttpCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
433
441
  export {};
434
- //# sourceMappingURL=http-crawler.d.ts.map