@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +1 -1
  2. package/configuration.d.ts +16 -47
  3. package/configuration.js +13 -25
  4. package/debug.js +4 -4
  5. package/errors.d.ts +28 -38
  6. package/errors.js +33 -47
  7. package/events/event_manager.d.ts +2 -2
  8. package/events/event_manager.js +7 -6
  9. package/events/index.d.ts +1 -0
  10. package/events/local_event_manager.d.ts +1 -8
  11. package/events/local_event_manager.js +13 -13
  12. package/events/system_info.d.ts +38 -0
  13. package/index.d.ts +2 -8
  14. package/index.js +4 -8
  15. package/internal.d.ts +8 -0
  16. package/internal.js +9 -0
  17. package/log.d.ts +10 -11
  18. package/log.js +52 -20
  19. package/memory-storage/memory-storage.d.ts +15 -18
  20. package/memory-storage/memory-storage.js +80 -58
  21. package/memory-storage/resource-clients/dataset.d.ts +1 -6
  22. package/memory-storage/resource-clients/dataset.js +23 -31
  23. package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
  24. package/memory-storage/resource-clients/key-value-store.js +43 -67
  25. package/memory-storage/resource-clients/request-queue.d.ts +1 -42
  26. package/memory-storage/resource-clients/request-queue.js +109 -117
  27. package/owned_or_injected.d.ts +1 -3
  28. package/owned_or_injected.js +17 -17
  29. package/package.json +17 -20
  30. package/proxy_configuration.d.ts +21 -26
  31. package/proxy_configuration.js +35 -25
  32. package/recoverable_state.d.ts +104 -47
  33. package/recoverable_state.js +199 -74
  34. package/request.d.ts +20 -107
  35. package/request.js +78 -244
  36. package/serialization.js +17 -16
  37. package/service_locator.d.ts +22 -10
  38. package/service_locator.js +59 -48
  39. package/storages/batched_adds.d.ts +37 -0
  40. package/storages/batched_adds.js +73 -0
  41. package/storages/dataset.d.ts +13 -8
  42. package/storages/dataset.js +149 -40
  43. package/storages/index.d.ts +4 -4
  44. package/storages/index.js +2 -4
  45. package/storages/key_value_store.d.ts +16 -35
  46. package/storages/key_value_store.js +223 -110
  47. package/storages/key_value_store_codec.js +6 -11
  48. package/storages/request_dedup_cache.d.ts +1 -4
  49. package/storages/request_dedup_cache.js +15 -15
  50. package/storages/request_list.d.ts +9 -104
  51. package/storages/request_list.js +236 -233
  52. package/storages/request_loader.d.ts +49 -18
  53. package/storages/request_loader.js +36 -1
  54. package/storages/request_manager.d.ts +86 -0
  55. package/storages/request_manager_tandem.d.ts +14 -38
  56. package/storages/request_manager_tandem.js +67 -64
  57. package/storages/request_queue.d.ts +23 -50
  58. package/storages/request_queue.js +371 -226
  59. package/storages/storage_instance_manager.d.ts +2 -4
  60. package/storages/storage_instance_manager.js +21 -21
  61. package/storages/storage_stats.d.ts +1 -1
  62. package/storages/storage_stats.js +4 -4
  63. package/storages/transaction.d.ts +270 -0
  64. package/storages/transaction.js +296 -0
  65. package/storages/utils.d.ts +6 -3
  66. package/storages/utils.js +11 -2
  67. package/system-info/runtime.js +7 -7
  68. package/url.d.ts +9 -0
  69. package/url.js +11 -0
  70. package/validators.d.ts +23 -25
  71. package/validators.js +14 -25
  72. package/autoscaling/autoscaled_pool.d.ts +0 -213
  73. package/autoscaling/autoscaled_pool.js +0 -378
  74. package/autoscaling/client_load_signal.d.ts +0 -59
  75. package/autoscaling/client_load_signal.js +0 -73
  76. package/autoscaling/concurrency_system.d.ts +0 -283
  77. package/autoscaling/concurrency_system.js +0 -350
  78. package/autoscaling/cpu_load_signal.d.ts +0 -44
  79. package/autoscaling/cpu_load_signal.js +0 -46
  80. package/autoscaling/event_loop_load_signal.d.ts +0 -54
  81. package/autoscaling/event_loop_load_signal.js +0 -60
  82. package/autoscaling/index.d.ts +0 -9
  83. package/autoscaling/index.js +0 -9
  84. package/autoscaling/load_signal.d.ts +0 -99
  85. package/autoscaling/load_signal.js +0 -103
  86. package/autoscaling/memory_load_signal.d.ts +0 -56
  87. package/autoscaling/memory_load_signal.js +0 -106
  88. package/autoscaling/snapshotter.d.ts +0 -87
  89. package/autoscaling/snapshotter.js +0 -67
  90. package/autoscaling/system_status.d.ts +0 -161
  91. package/autoscaling/system_status.js +0 -139
  92. package/autoscaling/weighted_avg.d.ts +0 -5
  93. package/autoscaling/weighted_avg.js +0 -14
  94. package/cookie_utils.d.ts +0 -44
  95. package/cookie_utils.js +0 -122
  96. package/crawlers/context_pipeline.d.ts +0 -70
  97. package/crawlers/context_pipeline.js +0 -122
  98. package/crawlers/crawler_commons.d.ts +0 -257
  99. package/crawlers/crawler_commons.js +0 -107
  100. package/crawlers/error_snapshotter.d.ts +0 -59
  101. package/crawlers/error_snapshotter.js +0 -117
  102. package/crawlers/error_tracker.d.ts +0 -54
  103. package/crawlers/error_tracker.js +0 -308
  104. package/crawlers/index.d.ts +0 -5
  105. package/crawlers/index.js +0 -5
  106. package/crawlers/internals/types.d.ts +0 -7
  107. package/crawlers/statistics.d.ts +0 -209
  108. package/crawlers/statistics.js +0 -350
  109. package/enqueue_links/enqueue_links.d.ts +0 -264
  110. package/enqueue_links/enqueue_links.js +0 -271
  111. package/enqueue_links/index.d.ts +0 -2
  112. package/enqueue_links/index.js +0 -2
  113. package/enqueue_links/shared.d.ts +0 -83
  114. package/enqueue_links/shared.js +0 -221
  115. package/router.d.ts +0 -309
  116. package/router.js +0 -309
  117. package/session_pool/consts.d.ts +0 -3
  118. package/session_pool/consts.js +0 -3
  119. package/session_pool/errors.d.ts +0 -7
  120. package/session_pool/errors.js +0 -11
  121. package/session_pool/fingerprint.d.ts +0 -9
  122. package/session_pool/fingerprint.js +0 -30
  123. package/session_pool/index.d.ts +0 -4
  124. package/session_pool/index.js +0 -4
  125. package/session_pool/session.d.ts +0 -161
  126. package/session_pool/session.js +0 -218
  127. package/session_pool/session_pool.d.ts +0 -246
  128. package/session_pool/session_pool.js +0 -386
  129. package/storages/access_checking.d.ts +0 -12
  130. package/storages/access_checking.js +0 -17
  131. package/storages/sitemap_request_loader.d.ts +0 -249
  132. package/storages/sitemap_request_loader.js +0 -432
  133. /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
package/router.d.ts DELETED
@@ -1,309 +0,0 @@
1
- import type { Awaitable, Dictionary } from '@crawlee/types';
2
- import type { StandardSchemaV1 } from '@standard-schema/spec';
3
- import type { CrawlingContext, LoadedRequest, RestrictedCrawlingContext, TypedContextAddRequests, TypedContextEnqueueLinks } from './crawlers/crawler_commons.js';
4
- import type { Request } from './request.js';
5
- /**
6
- * The key of the default route — the fallback handler registered via {@link Router.addDefaultHandler}.
7
- * Use it in a {@link RouteSchemas} map to register a schema that validates the `userData` of every request
8
- * that falls through to the default handler (i.e. whose label has no route of its own).
9
- */
10
- export declare const defaultRoute: unique symbol;
11
- /**
12
- * The crawling context received by a route handler, with `request.userData` narrowed to `UserData`, and
13
- * `addRequests`/`enqueueLinks` typed according to the router's route map (`Routes`) so that enqueuing a
14
- * request under a declared label requires the matching `userData` shape.
15
- */
16
- export type RouterHandlerContext<Context, UserData extends Dictionary, Routes extends Record<keyof Routes, Dictionary>> = Omit<Context, 'request' | 'addRequests' | 'enqueueLinks'> & {
17
- request: LoadedRequest<Request<UserData>>;
18
- addRequests: TypedContextAddRequests<Routes>;
19
- } & (Context extends {
20
- enqueueLinks: infer EnqueueLinks;
21
- } ? {
22
- enqueueLinks: TypedContextEnqueueLinks<EnqueueLinks, Routes>;
23
- } : {});
24
- /**
25
- * A map of request labels to a [Standard Schema](https://standardschema.dev) (Zod, Valibot, ArkType, …)
26
- * validating that label's `request.userData`. Pass it to {@link Router.create} or a `createXRouter`
27
- * factory to derive the per-label `request.userData` types *and* validate them at runtime. The optional
28
- * {@link defaultRoute} key registers a schema for requests handled by the default route.
29
- */
30
- export type RouteSchemas = Record<string, StandardSchemaV1> & {
31
- [defaultRoute]?: StandardSchemaV1;
32
- };
33
- /** Infers a label's `userData` type from its schema, falling back to a plain {@link Dictionary}. */
34
- type SchemaUserData<Schema extends StandardSchemaV1> = StandardSchemaV1.InferOutput<Schema> extends Dictionary ? StandardSchemaV1.InferOutput<Schema> : Dictionary;
35
- /**
36
- * Derives a route map (label → `userData` type) from a {@link RouteSchemas} map by inferring each schema's
37
- * output type. Outputs that are not object-shaped fall back to a plain {@link Dictionary}. The
38
- * {@link defaultRoute} schema is kept under its symbol key so {@link Router.addDefaultHandler} can pick it
39
- * up; string labels (the ones {@link Router.addHandler} and the crawler-level typing accept) ignore it.
40
- */
41
- export type RoutesFromSchemas<Schemas extends RouteSchemas> = {
42
- [Label in Extract<keyof Schemas, string>]: SchemaUserData<Schemas[Label]>;
43
- } & (Schemas extends {
44
- [defaultRoute]: StandardSchemaV1;
45
- } ? {
46
- [defaultRoute]: SchemaUserData<Schemas[typeof defaultRoute]>;
47
- } : {});
48
- /**
49
- * The `userData` type of the default route: inferred from the {@link defaultRoute} schema when the route map
50
- * carries one, otherwise the provided `Fallback`.
51
- */
52
- export type DefaultRouteUserData<Routes, Fallback extends Dictionary> = Routes extends {
53
- [defaultRoute]: infer DefaultUserData extends Dictionary;
54
- } ? DefaultUserData : Fallback;
55
- /**
56
- * Validates `userData` against a {@link RouteSchemas|Standard Schema}, returning the parsed (and coerced)
57
- * value. Throws a {@link RequestValidationError} when validation fails.
58
- * @internal
59
- */
60
- export declare function validateUserData(label: string | symbol, schema: StandardSchemaV1, userData: unknown): Promise<Dictionary>;
61
- /**
62
- * The set of labels accepted by {@link Router.addHandler}. When the router declares a concrete
63
- * route map (e.g. `{ PRODUCT: ...; CATEGORY: ... }`), only those labels (plus symbols) are
64
- * allowed — unknown labels become a compile-time error. When the map is left open (the default
65
- * `Record<string, ...>`), any string or symbol label is accepted, preserving the original behaviour.
66
- */
67
- export type RouterLabel<Routes extends Record<keyof Routes, Dictionary>> = string extends keyof Routes ? string | symbol : (keyof Routes & string) | symbol;
68
- export interface RouterHandler<Context extends Omit<RestrictedCrawlingContext, 'enqueueLinks'> = CrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>> extends Router<Context, Routes> {
69
- (ctx: Context): Awaitable<void>;
70
- }
71
- export type GetUserDataFromRequest<T> = T extends Request<infer Y> ? Y : never;
72
- /**
73
- * Per-route overrides, passed as the last argument of {@link Router.addHandler|`addHandler`} and
74
- * {@link Router.addDefaultHandler|`addDefaultHandler`}.
75
- */
76
- export interface RouteOptions {
77
- /**
78
- * Overrides the crawler's `requestHandlerTimeoutSecs` for this route only. Useful when one kind of page
79
- * needs markedly more time than the rest - a listing page behind an infinite scroll, say - and you do not
80
- * want to raise the timeout for every other page to accommodate it.
81
- *
82
- * Applies only to this route's handler. The navigation and the navigation hooks keep their own timeouts.
83
- */
84
- requestHandlerTimeoutSecs?: number;
85
- }
86
- export type RouterRoutes<Context, Routes extends Record<keyof Routes, Dictionary>> = {
87
- [Label in keyof Routes]: (ctx: Omit<Context, 'request'> & {
88
- request: Request<Routes[Label]>;
89
- }) => Awaitable<void>;
90
- };
91
- /**
92
- * Simple router that works based on request labels. This instance can then serve as a `requestHandler` of your crawler.
93
- *
94
- * ```ts
95
- * import { Router, CheerioCrawler, CheerioCrawlingContext } from 'crawlee';
96
- *
97
- * const router = Router.create<CheerioCrawlingContext>();
98
- *
99
- * // we can also use factory methods for specific crawling contexts, the above equals to:
100
- * // import { createCheerioRouter } from 'crawlee';
101
- * // const router = createCheerioRouter();
102
- *
103
- * router.addHandler('label-a', async (ctx) => {
104
- * ctx.log.info('...');
105
- * });
106
- * router.addDefaultHandler(async (ctx) => {
107
- * ctx.log.info('...');
108
- * });
109
- *
110
- * const crawler = new CheerioCrawler({
111
- * requestHandler: router,
112
- * });
113
- * await crawler.run();
114
- * ```
115
- *
116
- * Alternatively we can use the default router instance from crawler object:
117
- *
118
- * ```ts
119
- * import { CheerioCrawler } from 'crawlee';
120
- *
121
- * const crawler = new CheerioCrawler();
122
- *
123
- * crawler.router.addHandler('label-a', async (ctx) => {
124
- * ctx.log.info('...');
125
- * });
126
- * crawler.router.addDefaultHandler(async (ctx) => {
127
- * ctx.log.info('...');
128
- * });
129
- *
130
- * await crawler.run();
131
- * ```
132
- *
133
- * For convenience, we can also define the routes right when creating the router:
134
- *
135
- * ```ts
136
- * import { CheerioCrawler, createCheerioRouter } from 'crawlee';
137
- * const crawler = new CheerioCrawler({
138
- * requestHandler: createCheerioRouter({
139
- * 'label-a': async (ctx) => { ... },
140
- * 'label-b': async (ctx) => { ... },
141
- * })},
142
- * });
143
- * await crawler.run();
144
- * ```
145
- *
146
- * Middlewares are also supported via the `router.use` method. There can be multiple
147
- * middlewares for a single router, they will be executed sequentially in the same
148
- * order as they were registered.
149
- *
150
- * ```ts
151
- * crawler.router.use(async (ctx) => {
152
- * ctx.log.info('...');
153
- * });
154
- * ```
155
- *
156
- * To get `request.userData` typed per label, declare a route map and pass it as the second
157
- * type argument. The label passed to {@link Router.addHandler} then drives the type of
158
- * `request.userData`, and unknown labels are rejected at compile time:
159
- *
160
- * ```ts
161
- * import { createCheerioRouter, CheerioCrawlingContext } from 'crawlee';
162
- *
163
- * interface Routes {
164
- * PRODUCT: { sku: string; price: number };
165
- * CATEGORY: { categoryId: string };
166
- * }
167
- *
168
- * const router = createCheerioRouter<CheerioCrawlingContext, Routes>();
169
- *
170
- * router.addHandler('PRODUCT', async ({ request }) => {
171
- * request.userData.sku; // string
172
- * request.userData.price; // number
173
- * });
174
- *
175
- * router.addHandler('TYPO', async () => {}); // compile error: not a known label
176
- * ```
177
- *
178
- * Passing a [Standard Schema](https://standardschema.dev) per label instead of a plain type both infers the
179
- * `request.userData` types *and* validates them at runtime — when the request is handled, and when it is
180
- * added to the crawler (`crawler.addRequests`, `context.addRequests`, `enqueueLinks`). A failing request
181
- * throws a {@link RequestValidationError}.
182
- *
183
- * ```ts
184
- * import { z } from 'zod';
185
- * import { createCheerioRouter } from 'crawlee';
186
- *
187
- * const router = createCheerioRouter({
188
- * PRODUCT: z.object({ sku: z.string(), price: z.number() }),
189
- * CATEGORY: z.object({ categoryId: z.string() }),
190
- * });
191
- *
192
- * router.addHandler('PRODUCT', async ({ request }) => {
193
- * request.userData.price; // number, inferred from the schema and validated at runtime
194
- * });
195
- * ```
196
- *
197
- * A single route can take longer than the rest without raising the crawler-wide
198
- * `requestHandlerTimeoutSecs` for everything - pass a per-route timeout as the last argument:
199
- *
200
- * ```ts
201
- * // LIST pages scroll through a lot of content, DETAIL pages are quick
202
- * router.addHandler('LIST', async (ctx) => { ... }, { requestHandlerTimeoutSecs: 120 });
203
- * router.addHandler('DETAIL', async (ctx) => { ... }); // keeps the crawler's default
204
- * ```
205
- *
206
- * When the time a route needs is only apparent once it is already running, call
207
- * {@link CrawlingContext.extendTimeout|`context.extendTimeout`} from inside the handler:
208
- *
209
- * ```ts
210
- * router.addHandler('LIST', async ({ page, extendTimeout }) => {
211
- * const pageCount = await countPages(page);
212
- * extendTimeout(pageCount * 10); // ask for 10 more seconds per page
213
- * await scrapeAllPages(page);
214
- * });
215
- * ```
216
- */
217
- export declare class Router<Context extends Omit<RestrictedCrawlingContext, 'enqueueLinks'>, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>> {
218
- private readonly routes;
219
- private readonly schemas;
220
- private readonly timeouts;
221
- private readonly middlewares;
222
- /**
223
- * use Router.create() instead!
224
- * @ignore
225
- */
226
- private constructor();
227
- /**
228
- * Registers new route handler for given label. When the router declares a route map, the
229
- * `label` is restricted to the declared labels and `request.userData` is typed accordingly. Pass
230
- * {@link RouteOptions|`options`} to give this route its own `requestHandlerTimeoutSecs`,
231
- * overriding the crawler's default for requests with this label.
232
- */
233
- addHandler<Label extends keyof Routes & string>(label: Label, handler: (ctx: RouterHandlerContext<Context, Routes[Label], Routes>) => Awaitable<void>, options?: RouteOptions): void;
234
- /**
235
- * Registers new route handler for given label, explicitly typing `request.userData` via the
236
- * `UserData` type argument. Useful when the router has no declared route map (the open default)
237
- * and you want to type a single handler, or to register a handler under a `symbol` label.
238
- */
239
- addHandler<UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(label: RouterLabel<Routes>, handler: (ctx: RouterHandlerContext<Context, UserData, Routes>) => Awaitable<void>, options?: RouteOptions): void;
240
- /**
241
- * Registers default route handler. As a fallback it can receive any request (including labels not
242
- * declared in the route map). When the router was created with a {@link defaultRoute} schema,
243
- * `request.userData` is typed from it; otherwise it defaults to the context's (loosely typed) `userData`.
244
- * Pass an explicit `UserData` type argument to narrow it. Pass {@link RouteOptions|`options`} to give the
245
- * default route its own `requestHandlerTimeoutSecs`, overriding the crawler's default for requests that fall
246
- * through to it.
247
- */
248
- addDefaultHandler<UserData extends Dictionary = DefaultRouteUserData<Routes, GetUserDataFromRequest<Context['request']>>>(handler: (ctx: RouterHandlerContext<Context, UserData, Routes>) => Awaitable<void>, options?: RouteOptions): void;
249
- /**
250
- * Returns the {@link RouteSchemas|Standard Schema} registered for a label, if any. Used by the crawler
251
- * to validate `request.userData` when requests are added.
252
- * @internal
253
- */
254
- getSchema(label?: string | symbol): StandardSchemaV1 | undefined;
255
- /**
256
- * Registers a middleware that will be fired before the matching route handler.
257
- * Multiple middlewares can be registered, they will be fired in the same order.
258
- */
259
- use(middleware: (ctx: Context) => Awaitable<void>): void;
260
- /**
261
- * Returns the `requestHandlerTimeoutSecs` registered for a label, or `undefined` when the route did not
262
- * override it and the crawler's own timeout should apply. Falls back to the default route the same way
263
- * {@link Router.getHandler|`getHandler`} does, so a label with no route of its own inherits whatever
264
- * the default route asked for. Used by the crawler; not meant to be called directly.
265
- */
266
- getTimeoutSecs(label?: string | symbol): number | undefined;
267
- /**
268
- * The longest `requestHandlerTimeoutSecs` any route asked for, or `undefined` when no route overrides it.
269
- * The crawler needs an upper bound up front, before it knows which routes a run will actually hit.
270
- */
271
- getMaxTimeoutSecs(): number | undefined;
272
- /**
273
- * Returns route handler for given label. If no label is provided, the default request handler will be returned.
274
- */
275
- getHandler(label?: string | symbol): (ctx: Context) => Awaitable<void>;
276
- /**
277
- * Validates `request.userData` against the schema registered for its label (if any), replacing it with
278
- * the parsed value. Throws a {@link RequestValidationError} when validation fails.
279
- */
280
- private validateRequest;
281
- /**
282
- * Throws when the label already exists in our registry.
283
- */
284
- private validate;
285
- /**
286
- * Creates new router instance. This instance can then serve as a `requestHandler` of your crawler.
287
- *
288
- * ```ts
289
- * import { Router, CheerioCrawler, CheerioCrawlingContext } from 'crawlee';
290
- *
291
- * const router = Router.create<CheerioCrawlingContext>();
292
- * router.addHandler('label-a', async (ctx) => {
293
- * ctx.log.info('...');
294
- * });
295
- * router.addDefaultHandler(async (ctx) => {
296
- * ctx.log.info('...');
297
- * });
298
- *
299
- * const crawler = new CheerioCrawler({
300
- * requestHandler: router,
301
- * });
302
- * await crawler.run();
303
- * ```
304
- */
305
- static create<Context extends Omit<RestrictedCrawlingContext, 'enqueueLinks'> = CrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
306
- static create<Context extends Omit<RestrictedCrawlingContext, 'enqueueLinks'> = CrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
307
- static create<Context extends Omit<RestrictedCrawlingContext, 'enqueueLinks'> = CrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
308
- }
309
- export {};
package/router.js DELETED
@@ -1,309 +0,0 @@
1
- import { MissingRouteError, RequestValidationError } from './errors.js';
2
- /**
3
- * The key of the default route — the fallback handler registered via {@link Router.addDefaultHandler}.
4
- * Use it in a {@link RouteSchemas} map to register a schema that validates the `userData` of every request
5
- * that falls through to the default handler (i.e. whose label has no route of its own).
6
- */
7
- export const defaultRoute = Symbol('default-route');
8
- /** Whether a validation issue points at the top-level `label` key. */
9
- function isLabelIssue(issue) {
10
- if (issue.path?.length !== 1) {
11
- return false;
12
- }
13
- const [segment] = issue.path;
14
- return (typeof segment === 'object' ? segment.key : segment) === 'label';
15
- }
16
- /**
17
- * Validates `userData` against a {@link RouteSchemas|Standard Schema}, returning the parsed (and coerced)
18
- * value. Throws a {@link RequestValidationError} when validation fails.
19
- * @internal
20
- */
21
- export async function validateUserData(label, schema, userData) {
22
- const { label: _label, ...rest } = (userData ?? {});
23
- // `label` is a Crawlee-managed key that lives inside `userData`, so validating it is opt-in: we validate
24
- // without it first, letting schemas that don't describe it pass (including `.strict()` ones). A schema that
25
- // *does* declare `label` reports an issue for the now-missing key — so we re-validate with it included,
26
- // honouring the declaration. Unlike `userData.__crawlee`, `label` is enumerable, so schemas do see it.
27
- let result = await schema['~standard'].validate(rest);
28
- if (result.issues?.some(isLabelIssue)) {
29
- result = await schema['~standard'].validate({ ...rest, label });
30
- }
31
- if (result.issues) {
32
- throw new RequestValidationError(label, result.issues);
33
- }
34
- // Restore the label so it survives schemas that strip undeclared keys.
35
- return { ...result.value, label };
36
- }
37
- /**
38
- * Simple router that works based on request labels. This instance can then serve as a `requestHandler` of your crawler.
39
- *
40
- * ```ts
41
- * import { Router, CheerioCrawler, CheerioCrawlingContext } from 'crawlee';
42
- *
43
- * const router = Router.create<CheerioCrawlingContext>();
44
- *
45
- * // we can also use factory methods for specific crawling contexts, the above equals to:
46
- * // import { createCheerioRouter } from 'crawlee';
47
- * // const router = createCheerioRouter();
48
- *
49
- * router.addHandler('label-a', async (ctx) => {
50
- * ctx.log.info('...');
51
- * });
52
- * router.addDefaultHandler(async (ctx) => {
53
- * ctx.log.info('...');
54
- * });
55
- *
56
- * const crawler = new CheerioCrawler({
57
- * requestHandler: router,
58
- * });
59
- * await crawler.run();
60
- * ```
61
- *
62
- * Alternatively we can use the default router instance from crawler object:
63
- *
64
- * ```ts
65
- * import { CheerioCrawler } from 'crawlee';
66
- *
67
- * const crawler = new CheerioCrawler();
68
- *
69
- * crawler.router.addHandler('label-a', async (ctx) => {
70
- * ctx.log.info('...');
71
- * });
72
- * crawler.router.addDefaultHandler(async (ctx) => {
73
- * ctx.log.info('...');
74
- * });
75
- *
76
- * await crawler.run();
77
- * ```
78
- *
79
- * For convenience, we can also define the routes right when creating the router:
80
- *
81
- * ```ts
82
- * import { CheerioCrawler, createCheerioRouter } from 'crawlee';
83
- * const crawler = new CheerioCrawler({
84
- * requestHandler: createCheerioRouter({
85
- * 'label-a': async (ctx) => { ... },
86
- * 'label-b': async (ctx) => { ... },
87
- * })},
88
- * });
89
- * await crawler.run();
90
- * ```
91
- *
92
- * Middlewares are also supported via the `router.use` method. There can be multiple
93
- * middlewares for a single router, they will be executed sequentially in the same
94
- * order as they were registered.
95
- *
96
- * ```ts
97
- * crawler.router.use(async (ctx) => {
98
- * ctx.log.info('...');
99
- * });
100
- * ```
101
- *
102
- * To get `request.userData` typed per label, declare a route map and pass it as the second
103
- * type argument. The label passed to {@link Router.addHandler} then drives the type of
104
- * `request.userData`, and unknown labels are rejected at compile time:
105
- *
106
- * ```ts
107
- * import { createCheerioRouter, CheerioCrawlingContext } from 'crawlee';
108
- *
109
- * interface Routes {
110
- * PRODUCT: { sku: string; price: number };
111
- * CATEGORY: { categoryId: string };
112
- * }
113
- *
114
- * const router = createCheerioRouter<CheerioCrawlingContext, Routes>();
115
- *
116
- * router.addHandler('PRODUCT', async ({ request }) => {
117
- * request.userData.sku; // string
118
- * request.userData.price; // number
119
- * });
120
- *
121
- * router.addHandler('TYPO', async () => {}); // compile error: not a known label
122
- * ```
123
- *
124
- * Passing a [Standard Schema](https://standardschema.dev) per label instead of a plain type both infers the
125
- * `request.userData` types *and* validates them at runtime — when the request is handled, and when it is
126
- * added to the crawler (`crawler.addRequests`, `context.addRequests`, `enqueueLinks`). A failing request
127
- * throws a {@link RequestValidationError}.
128
- *
129
- * ```ts
130
- * import { z } from 'zod';
131
- * import { createCheerioRouter } from 'crawlee';
132
- *
133
- * const router = createCheerioRouter({
134
- * PRODUCT: z.object({ sku: z.string(), price: z.number() }),
135
- * CATEGORY: z.object({ categoryId: z.string() }),
136
- * });
137
- *
138
- * router.addHandler('PRODUCT', async ({ request }) => {
139
- * request.userData.price; // number, inferred from the schema and validated at runtime
140
- * });
141
- * ```
142
- *
143
- * A single route can take longer than the rest without raising the crawler-wide
144
- * `requestHandlerTimeoutSecs` for everything - pass a per-route timeout as the last argument:
145
- *
146
- * ```ts
147
- * // LIST pages scroll through a lot of content, DETAIL pages are quick
148
- * router.addHandler('LIST', async (ctx) => { ... }, { requestHandlerTimeoutSecs: 120 });
149
- * router.addHandler('DETAIL', async (ctx) => { ... }); // keeps the crawler's default
150
- * ```
151
- *
152
- * When the time a route needs is only apparent once it is already running, call
153
- * {@link CrawlingContext.extendTimeout|`context.extendTimeout`} from inside the handler:
154
- *
155
- * ```ts
156
- * router.addHandler('LIST', async ({ page, extendTimeout }) => {
157
- * const pageCount = await countPages(page);
158
- * extendTimeout(pageCount * 10); // ask for 10 more seconds per page
159
- * await scrapeAllPages(page);
160
- * });
161
- * ```
162
- */
163
- export class Router {
164
- routes = new Map();
165
- schemas = new Map();
166
- timeouts = new Map();
167
- middlewares = [];
168
- /**
169
- * use Router.create() instead!
170
- * @ignore
171
- */
172
- constructor() { }
173
- addHandler(label, handler, options = {}) {
174
- this.validate(label);
175
- this.routes.set(label, handler);
176
- if (options.requestHandlerTimeoutSecs !== undefined) {
177
- this.timeouts.set(label, options.requestHandlerTimeoutSecs);
178
- }
179
- }
180
- /**
181
- * Registers default route handler. As a fallback it can receive any request (including labels not
182
- * declared in the route map). When the router was created with a {@link defaultRoute} schema,
183
- * `request.userData` is typed from it; otherwise it defaults to the context's (loosely typed) `userData`.
184
- * Pass an explicit `UserData` type argument to narrow it. Pass {@link RouteOptions|`options`} to give the
185
- * default route its own `requestHandlerTimeoutSecs`, overriding the crawler's default for requests that fall
186
- * through to it.
187
- */
188
- addDefaultHandler(handler, options = {}) {
189
- this.validate(defaultRoute);
190
- this.routes.set(defaultRoute, handler);
191
- if (options.requestHandlerTimeoutSecs !== undefined) {
192
- this.timeouts.set(defaultRoute, options.requestHandlerTimeoutSecs);
193
- }
194
- }
195
- /**
196
- * Returns the {@link RouteSchemas|Standard Schema} registered for a label, if any. Used by the crawler
197
- * to validate `request.userData` when requests are added.
198
- * @internal
199
- */
200
- getSchema(label) {
201
- if (label != null) {
202
- const schema = this.schemas.get(label);
203
- if (schema) {
204
- return schema;
205
- }
206
- // A label with its own route is fully specified; don't fall back to the default-route schema.
207
- if (this.routes.has(label)) {
208
- return undefined;
209
- }
210
- }
211
- // Requests with no route of their own fall through to the default handler, so validate their
212
- // `userData` against the default-route schema, if one was registered.
213
- return this.schemas.get(defaultRoute);
214
- }
215
- /**
216
- * Registers a middleware that will be fired before the matching route handler.
217
- * Multiple middlewares can be registered, they will be fired in the same order.
218
- */
219
- use(middleware) {
220
- this.middlewares.push(middleware);
221
- }
222
- /**
223
- * Returns the `requestHandlerTimeoutSecs` registered for a label, or `undefined` when the route did not
224
- * override it and the crawler's own timeout should apply. Falls back to the default route the same way
225
- * {@link Router.getHandler|`getHandler`} does, so a label with no route of its own inherits whatever
226
- * the default route asked for. Used by the crawler; not meant to be called directly.
227
- */
228
- getTimeoutSecs(label) {
229
- if (label && this.routes.has(label)) {
230
- return this.timeouts.get(label);
231
- }
232
- return this.timeouts.get(defaultRoute);
233
- }
234
- /**
235
- * The longest `requestHandlerTimeoutSecs` any route asked for, or `undefined` when no route overrides it.
236
- * The crawler needs an upper bound up front, before it knows which routes a run will actually hit.
237
- */
238
- getMaxTimeoutSecs() {
239
- return this.timeouts.size > 0 ? Math.max(...this.timeouts.values()) : undefined;
240
- }
241
- /**
242
- * Returns route handler for given label. If no label is provided, the default request handler will be returned.
243
- */
244
- getHandler(label) {
245
- if (label && this.routes.has(label)) {
246
- return this.routes.get(label);
247
- }
248
- if (this.routes.has(defaultRoute)) {
249
- return this.routes.get(defaultRoute);
250
- }
251
- throw new MissingRouteError(`Route not found for label '${String(label)}'.` +
252
- ' You must set up a route for this label or a default route.' +
253
- ' Use `requestHandler`, `router.addHandler` or `router.addDefaultHandler`.');
254
- }
255
- /**
256
- * Validates `request.userData` against the schema registered for its label (if any), replacing it with
257
- * the parsed value. Throws a {@link RequestValidationError} when validation fails.
258
- */
259
- async validateRequest(context) {
260
- const label = context.request.label;
261
- const schema = this.getSchema(label);
262
- if (schema) {
263
- context.request.userData = (await validateUserData(label, schema, context.request.userData));
264
- }
265
- }
266
- /**
267
- * Throws when the label already exists in our registry.
268
- */
269
- validate(label) {
270
- if (this.routes.has(label)) {
271
- const message = label === defaultRoute
272
- ? `Default route is already defined!`
273
- : `Route for label '${String(label)}' is already defined!`;
274
- throw new Error(message);
275
- }
276
- }
277
- static create(routesOrSchemas) {
278
- const router = new Router();
279
- const obj = Object.create(Function.prototype);
280
- obj.addHandler = router.addHandler.bind(router);
281
- obj.addDefaultHandler = router.addDefaultHandler.bind(router);
282
- obj.getSchema = router.getSchema.bind(router);
283
- obj.getHandler = router.getHandler.bind(router);
284
- obj.getTimeoutSecs = router.getTimeoutSecs.bind(router);
285
- obj.getMaxTimeoutSecs = router.getMaxTimeoutSecs.bind(router);
286
- obj.use = router.use.bind(router);
287
- // `Reflect.ownKeys` (unlike `Object.entries`) also yields the `defaultRoute` symbol key.
288
- for (const label of Reflect.ownKeys(routesOrSchemas ?? {})) {
289
- const value = routesOrSchemas[label];
290
- if (typeof value === 'function') {
291
- router.addHandler(label, value);
292
- }
293
- else {
294
- router.schemas.set(label, value);
295
- }
296
- }
297
- const func = async function (context) {
298
- const { url, loadedUrl, label } = context.request;
299
- context.log.debug('Page opened.', { label, url: loadedUrl ?? url });
300
- await router.validateRequest(context);
301
- for (const middleware of router.middlewares) {
302
- await middleware(context);
303
- }
304
- return router.getHandler(label)(context);
305
- };
306
- Object.setPrototypeOf(func, obj);
307
- return func;
308
- }
309
- }
@@ -1,3 +0,0 @@
1
- export declare const BLOCKED_STATUS_CODES: number[];
2
- export declare const PERSIST_STATE_KEY = "CRAWLEE_SESSION_POOL_STATE";
3
- export declare const MAX_POOL_SIZE = 1000;
@@ -1,3 +0,0 @@
1
- export const BLOCKED_STATUS_CODES = [401, 403, 429];
2
- export const PERSIST_STATE_KEY = 'CRAWLEE_SESSION_POOL_STATE';
3
- export const MAX_POOL_SIZE = 1000;
@@ -1,7 +0,0 @@
1
- /**
2
- * @ignore
3
- */
4
- export declare class CookieParseError extends Error {
5
- readonly cookieHeaderString: unknown;
6
- constructor(cookieHeaderString: unknown);
7
- }
@@ -1,11 +0,0 @@
1
- /**
2
- * @ignore
3
- */
4
- export class CookieParseError extends Error {
5
- cookieHeaderString;
6
- constructor(cookieHeaderString) {
7
- super(`Could not parse cookie header string: ${cookieHeaderString}`);
8
- this.cookieHeaderString = cookieHeaderString;
9
- Error.captureStackTrace(this, CookieParseError);
10
- }
11
- }
@@ -1,9 +0,0 @@
1
- import type { SessionFingerprint } from '@crawlee/types';
2
- /**
3
- * Build a {@link SessionFingerprint} whose `platform` matches the host OS
4
- * and whose `browser`/`device` are randomized within the realistic profiles for
5
- * that platform. Used by {@link SessionPool} as the default fingerprint for
6
- * freshly created sessions; callers can override by passing their own
7
- * `fingerprint` in `sessionOptions`.
8
- */
9
- export declare function createDefaultSessionFingerprint(): SessionFingerprint;