@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/configuration.d.ts +16 -47
- package/configuration.js +13 -25
- package/debug.js +4 -4
- package/errors.d.ts +28 -38
- package/errors.js +33 -47
- package/events/event_manager.d.ts +2 -2
- package/events/event_manager.js +7 -6
- package/events/index.d.ts +1 -0
- package/events/local_event_manager.d.ts +1 -8
- package/events/local_event_manager.js +13 -13
- package/events/system_info.d.ts +38 -0
- package/index.d.ts +2 -8
- package/index.js +4 -8
- package/internal.d.ts +8 -0
- package/internal.js +9 -0
- package/log.d.ts +10 -11
- package/log.js +52 -20
- package/memory-storage/memory-storage.d.ts +15 -18
- package/memory-storage/memory-storage.js +80 -58
- package/memory-storage/resource-clients/dataset.d.ts +1 -6
- package/memory-storage/resource-clients/dataset.js +23 -31
- package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
- package/memory-storage/resource-clients/key-value-store.js +43 -67
- package/memory-storage/resource-clients/request-queue.d.ts +1 -42
- package/memory-storage/resource-clients/request-queue.js +109 -117
- package/owned_or_injected.d.ts +1 -3
- package/owned_or_injected.js +17 -17
- package/package.json +17 -20
- package/proxy_configuration.d.ts +21 -26
- package/proxy_configuration.js +35 -25
- package/recoverable_state.d.ts +104 -47
- package/recoverable_state.js +199 -74
- package/request.d.ts +20 -107
- package/request.js +78 -244
- package/serialization.js +17 -16
- package/service_locator.d.ts +22 -10
- package/service_locator.js +59 -48
- package/storages/batched_adds.d.ts +37 -0
- package/storages/batched_adds.js +73 -0
- package/storages/dataset.d.ts +13 -8
- package/storages/dataset.js +149 -40
- package/storages/index.d.ts +4 -4
- package/storages/index.js +2 -4
- package/storages/key_value_store.d.ts +16 -35
- package/storages/key_value_store.js +223 -110
- package/storages/key_value_store_codec.js +6 -11
- package/storages/request_dedup_cache.d.ts +1 -4
- package/storages/request_dedup_cache.js +15 -15
- package/storages/request_list.d.ts +9 -104
- package/storages/request_list.js +236 -233
- package/storages/request_loader.d.ts +49 -18
- package/storages/request_loader.js +36 -1
- package/storages/request_manager.d.ts +86 -0
- package/storages/request_manager_tandem.d.ts +14 -38
- package/storages/request_manager_tandem.js +67 -64
- package/storages/request_queue.d.ts +23 -50
- package/storages/request_queue.js +371 -226
- package/storages/storage_instance_manager.d.ts +2 -4
- package/storages/storage_instance_manager.js +21 -21
- package/storages/storage_stats.d.ts +1 -1
- package/storages/storage_stats.js +4 -4
- package/storages/transaction.d.ts +270 -0
- package/storages/transaction.js +296 -0
- package/storages/utils.d.ts +6 -3
- package/storages/utils.js +11 -2
- package/system-info/runtime.js +7 -7
- package/url.d.ts +9 -0
- package/url.js +11 -0
- package/validators.d.ts +23 -25
- package/validators.js +14 -25
- package/autoscaling/autoscaled_pool.d.ts +0 -213
- package/autoscaling/autoscaled_pool.js +0 -378
- package/autoscaling/client_load_signal.d.ts +0 -59
- package/autoscaling/client_load_signal.js +0 -73
- package/autoscaling/concurrency_system.d.ts +0 -283
- package/autoscaling/concurrency_system.js +0 -350
- package/autoscaling/cpu_load_signal.d.ts +0 -44
- package/autoscaling/cpu_load_signal.js +0 -46
- package/autoscaling/event_loop_load_signal.d.ts +0 -54
- package/autoscaling/event_loop_load_signal.js +0 -60
- package/autoscaling/index.d.ts +0 -9
- package/autoscaling/index.js +0 -9
- package/autoscaling/load_signal.d.ts +0 -99
- package/autoscaling/load_signal.js +0 -103
- package/autoscaling/memory_load_signal.d.ts +0 -56
- package/autoscaling/memory_load_signal.js +0 -106
- package/autoscaling/snapshotter.d.ts +0 -87
- package/autoscaling/snapshotter.js +0 -67
- package/autoscaling/system_status.d.ts +0 -161
- package/autoscaling/system_status.js +0 -139
- package/autoscaling/weighted_avg.d.ts +0 -5
- package/autoscaling/weighted_avg.js +0 -14
- package/cookie_utils.d.ts +0 -44
- package/cookie_utils.js +0 -122
- package/crawlers/context_pipeline.d.ts +0 -70
- package/crawlers/context_pipeline.js +0 -122
- package/crawlers/crawler_commons.d.ts +0 -257
- package/crawlers/crawler_commons.js +0 -107
- package/crawlers/error_snapshotter.d.ts +0 -59
- package/crawlers/error_snapshotter.js +0 -117
- package/crawlers/error_tracker.d.ts +0 -54
- package/crawlers/error_tracker.js +0 -308
- package/crawlers/index.d.ts +0 -5
- package/crawlers/index.js +0 -5
- package/crawlers/internals/types.d.ts +0 -7
- package/crawlers/statistics.d.ts +0 -209
- package/crawlers/statistics.js +0 -350
- package/enqueue_links/enqueue_links.d.ts +0 -264
- package/enqueue_links/enqueue_links.js +0 -271
- package/enqueue_links/index.d.ts +0 -2
- package/enqueue_links/index.js +0 -2
- package/enqueue_links/shared.d.ts +0 -83
- package/enqueue_links/shared.js +0 -221
- package/router.d.ts +0 -309
- package/router.js +0 -309
- package/session_pool/consts.d.ts +0 -3
- package/session_pool/consts.js +0 -3
- package/session_pool/errors.d.ts +0 -7
- package/session_pool/errors.js +0 -11
- package/session_pool/fingerprint.d.ts +0 -9
- package/session_pool/fingerprint.js +0 -30
- package/session_pool/index.d.ts +0 -4
- package/session_pool/index.js +0 -4
- package/session_pool/session.d.ts +0 -161
- package/session_pool/session.js +0 -218
- package/session_pool/session_pool.d.ts +0 -246
- package/session_pool/session_pool.js +0 -386
- package/storages/access_checking.d.ts +0 -12
- package/storages/access_checking.js +0 -17
- package/storages/sitemap_request_loader.d.ts +0 -249
- package/storages/sitemap_request_loader.js +0 -432
- /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
|
@@ -1,257 +0,0 @@
|
|
|
1
|
-
import type { Dictionary, HttpRequestOptions, ISession, ProxyInfo, SendRequestOptions } from '@crawlee/types';
|
|
2
|
-
import type { ReadonlyDeep, SetRequired } from 'type-fest';
|
|
3
|
-
import type { Configuration } from '../configuration.js';
|
|
4
|
-
import type { EnqueueLinksOptions } from '../enqueue_links/enqueue_links.js';
|
|
5
|
-
import type { CrawleeLogger } from '../log.js';
|
|
6
|
-
import type { Request, RequestOptions, Source } from '../request.js';
|
|
7
|
-
import type { StorageIdentifier } from '../storages/storage_instance_manager.js';
|
|
8
|
-
import type { Dataset } from '../storages/dataset.js';
|
|
9
|
-
import { KeyValueStore, type RecordOptions } from '../storages/key_value_store.js';
|
|
10
|
-
import type { RequestQueueOperationOptions } from '../storages/request_queue.js';
|
|
11
|
-
/** @internal */
|
|
12
|
-
export type IsAny<T> = 0 extends 1 & T ? true : false;
|
|
13
|
-
/**
|
|
14
|
-
* A request input (URL string, request-options object, or {@link Request}) whose `userData` is typed
|
|
15
|
-
* according to its `label`, based on a router's route map.
|
|
16
|
-
*
|
|
17
|
-
* When the route map is open (the default `Record<string, ...>`), this is just the regular loose
|
|
18
|
-
* {@link Source} input. When the map declares concrete labels, providing a `label` requires the matching
|
|
19
|
-
* `userData` shape and rejects labels not present in the map; unlabeled requests keep loose `userData`.
|
|
20
|
-
*/
|
|
21
|
-
export type LabeledSource<Routes extends Record<keyof Routes, Dictionary>> = string extends keyof Routes ? string | Source : string | Request | ({
|
|
22
|
-
requestsFromUrl?: string;
|
|
23
|
-
regex?: RegExp;
|
|
24
|
-
} & ({
|
|
25
|
-
[Label in keyof Routes & string]: Omit<Partial<RequestOptions<Routes[Label]>>, 'label'> & {
|
|
26
|
-
label: Label;
|
|
27
|
-
};
|
|
28
|
-
}[keyof Routes & string] | (Omit<Partial<RequestOptions>, 'label'> & {
|
|
29
|
-
label?: undefined;
|
|
30
|
-
})));
|
|
31
|
-
/**
|
|
32
|
-
* The iterable/array of {@link LabeledSource} inputs accepted by the label-aware `addRequests`/`run`
|
|
33
|
-
* methods of a crawler bound to a typed router.
|
|
34
|
-
* @internal
|
|
35
|
-
*/
|
|
36
|
-
export type TypedRequestsLike<Routes extends Record<keyof Routes, Dictionary>> = AsyncIterable<LabeledSource<Routes>> | Iterable<LabeledSource<Routes>> | LabeledSource<Routes>[];
|
|
37
|
-
/**
|
|
38
|
-
* The label-aware `addRequests` method signature exposed on a request handler's context when the crawler is
|
|
39
|
-
* bound to a typed router. Mirrors {@link RestrictedCrawlingContext.addRequests} with typed sources.
|
|
40
|
-
*/
|
|
41
|
-
export type TypedContextAddRequests<Routes extends Record<keyof Routes, Dictionary>> = (requestsLike: ReadonlyDeep<LabeledSource<Routes>[]>, options?: ReadonlyDeep<RequestQueueOperationOptions>) => Promise<void>;
|
|
42
|
-
/**
|
|
43
|
-
* An `enqueueLinks`-options object with its `label`/`userData` retyped according to a router's route map: a
|
|
44
|
-
* declared `label` requires the matching `userData` shape (unknown labels are rejected), while unlabeled
|
|
45
|
-
* calls keep loose `userData`. Returns the options unchanged when the route map is open (the default).
|
|
46
|
-
*/
|
|
47
|
-
type TypedEnqueueLinksOptions<Options, Routes extends Record<keyof Routes, Dictionary>> = string extends keyof Routes ? Options : Omit<Options, 'label' | 'userData'> & ({
|
|
48
|
-
[Label in keyof Routes & string]: {
|
|
49
|
-
label: Label;
|
|
50
|
-
userData?: Routes[Label];
|
|
51
|
-
};
|
|
52
|
-
}[keyof Routes & string] | {
|
|
53
|
-
label?: undefined;
|
|
54
|
-
userData?: Dictionary;
|
|
55
|
-
});
|
|
56
|
-
/**
|
|
57
|
-
* Transforms a context's existing `enqueueLinks` method so that the `label`/`userData` in its options follow
|
|
58
|
-
* the router's route map, while preserving everything else about the signature (argument optionality and
|
|
59
|
-
* return type, which differ between crawler types).
|
|
60
|
-
*/
|
|
61
|
-
export type TypedContextEnqueueLinks<EnqueueLinks, Routes extends Record<keyof Routes, Dictionary>> = EnqueueLinks extends (options?: infer Options) => infer Result ? (options?: TypedEnqueueLinksOptions<Options, Routes>) => Result : EnqueueLinks extends (options: infer Options) => infer Result ? (options: TypedEnqueueLinksOptions<Options, Routes>) => Result : EnqueueLinks;
|
|
62
|
-
export type WithRequired<T, K extends keyof T> = T & {
|
|
63
|
-
[P in K]-?: T[P];
|
|
64
|
-
};
|
|
65
|
-
export type LoadedRequest<R extends Request> = WithRequired<R, 'id' | 'loadedUrl'>;
|
|
66
|
-
/** @internal */
|
|
67
|
-
export type LoadedContext<Context extends RestrictedCrawlingContext> = IsAny<Context> extends true ? Context : {
|
|
68
|
-
request: LoadedRequest<Context['request']>;
|
|
69
|
-
} & Omit<Context, 'request'>;
|
|
70
|
-
export interface RestrictedCrawlingContext<UserData extends Dictionary = Dictionary> {
|
|
71
|
-
id: string;
|
|
72
|
-
session: ISession;
|
|
73
|
-
/**
|
|
74
|
-
* An object with information about currently used proxy by the crawler
|
|
75
|
-
* and configured by the {@link ProxyConfiguration} class.
|
|
76
|
-
*/
|
|
77
|
-
proxyInfo?: ProxyInfo;
|
|
78
|
-
/**
|
|
79
|
-
* The original {@link Request} object.
|
|
80
|
-
*/
|
|
81
|
-
request: Request<UserData>;
|
|
82
|
-
/**
|
|
83
|
-
* This function allows you to push data to a {@link Dataset} specified by name, or the one currently used by the crawler.
|
|
84
|
-
*
|
|
85
|
-
* Shortcut for `crawler.pushData()`.
|
|
86
|
-
*
|
|
87
|
-
* @param [data] Data to be pushed to the default dataset.
|
|
88
|
-
*/
|
|
89
|
-
pushData(data: ReadonlyDeep<Parameters<Dataset['pushData']>[0]>, datasetIdentifier?: string | StorageIdentifier): Promise<void>;
|
|
90
|
-
/**
|
|
91
|
-
* This function automatically finds and enqueues links from the current page, adding them to the {@link RequestQueue}
|
|
92
|
-
* currently used by the crawler.
|
|
93
|
-
*
|
|
94
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of globs or regular expressions
|
|
95
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
96
|
-
*
|
|
97
|
-
* Check out the [Crawl a website with relative links](https://crawlee.dev/js/docs/examples/crawl-relative-links) example
|
|
98
|
-
* for more details regarding its usage.
|
|
99
|
-
*
|
|
100
|
-
* **Example usage**
|
|
101
|
-
*
|
|
102
|
-
* ```ts
|
|
103
|
-
* async requestHandler({ enqueueLinks }) {
|
|
104
|
-
* await enqueueLinks({
|
|
105
|
-
* globs: [
|
|
106
|
-
* 'https://www.example.com/handbags/*',
|
|
107
|
-
* ],
|
|
108
|
-
* });
|
|
109
|
-
* },
|
|
110
|
-
* ```
|
|
111
|
-
*
|
|
112
|
-
* @param [options] All `enqueueLinks()` parameters are passed via an options object.
|
|
113
|
-
*/
|
|
114
|
-
enqueueLinks: (options: ReadonlyDeep<Omit<SetRequired<EnqueueLinksOptions, 'urls'>, 'requestManager' | 'robotsTxtFile'>>) => Promise<unknown>;
|
|
115
|
-
/**
|
|
116
|
-
* Add requests directly to the request queue.
|
|
117
|
-
*
|
|
118
|
-
* @param requests The requests to add
|
|
119
|
-
* @param options Options for the request queue
|
|
120
|
-
*/
|
|
121
|
-
addRequests: (requestsLike: ReadonlyDeep<(string | Source)[]>, options?: ReadonlyDeep<RequestQueueOperationOptions>) => Promise<void>;
|
|
122
|
-
/**
|
|
123
|
-
* Returns the state - a piece of mutable persistent data shared across all the request handler runs.
|
|
124
|
-
*/
|
|
125
|
-
useState: <State extends Dictionary = Dictionary>(defaultValue?: State) => Promise<State>;
|
|
126
|
-
/**
|
|
127
|
-
* Get a key-value store with given name or id, or the default one for the crawler.
|
|
128
|
-
*/
|
|
129
|
-
getKeyValueStore: (identifier?: string | StorageIdentifier) => Promise<Pick<KeyValueStore, 'id' | 'name' | 'getValue' | 'getAutoSavedValue' | 'setValue' | 'getPublicUrl'>>;
|
|
130
|
-
/**
|
|
131
|
-
* A preconfigured logger for the request handler.
|
|
132
|
-
*/
|
|
133
|
-
log: CrawleeLogger;
|
|
134
|
-
}
|
|
135
|
-
export interface CrawlingContext<UserData extends Dictionary = Dictionary> extends RestrictedCrawlingContext<UserData> {
|
|
136
|
-
/**
|
|
137
|
-
* This function automatically finds and enqueues links from the current page, adding them to the {@link RequestQueue}
|
|
138
|
-
* currently used by the crawler.
|
|
139
|
-
*
|
|
140
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of globs or regular expressions
|
|
141
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
142
|
-
*
|
|
143
|
-
* Check out the [Crawl a website with relative links](https://crawlee.dev/js/docs/examples/crawl-relative-links) example
|
|
144
|
-
* for more details regarding its usage.
|
|
145
|
-
*
|
|
146
|
-
* **Example usage**
|
|
147
|
-
*
|
|
148
|
-
* ```ts
|
|
149
|
-
* async requestHandler({ enqueueLinks }) {
|
|
150
|
-
* await enqueueLinks({
|
|
151
|
-
* globs: [
|
|
152
|
-
* 'https://www.example.com/handbags/*',
|
|
153
|
-
* ],
|
|
154
|
-
* });
|
|
155
|
-
* },
|
|
156
|
-
* ```
|
|
157
|
-
*
|
|
158
|
-
* @param [options] All `enqueueLinks()` parameters are passed via an options object.
|
|
159
|
-
* @returns Promise that resolves to {@link BatchAddRequestsResult} object.
|
|
160
|
-
*/
|
|
161
|
-
enqueueLinks(options: ReadonlyDeep<Omit<SetRequired<EnqueueLinksOptions, 'urls'>, 'requestManager' | 'robotsTxtFile'>> & Pick<EnqueueLinksOptions, 'requestManager' | 'robotsTxtFile'>): Promise<unknown>;
|
|
162
|
-
/**
|
|
163
|
-
* Fires HTTP request via the internal HTTP client, allowing to override the request options on the fly.
|
|
164
|
-
*
|
|
165
|
-
* This is handy when you work with a browser crawler but want to execute some requests outside it (e.g. API requests).
|
|
166
|
-
* Check the [Skipping navigations for certain requests](https://crawlee.dev/js/docs/examples/skip-navigation) example for
|
|
167
|
-
* more detailed explanation of how to do that.
|
|
168
|
-
*
|
|
169
|
-
* ```ts
|
|
170
|
-
* async requestHandler({ sendRequest }) {
|
|
171
|
-
* const { body } = await sendRequest({
|
|
172
|
-
* // override headers only
|
|
173
|
-
* headers: { ... },
|
|
174
|
-
* });
|
|
175
|
-
* },
|
|
176
|
-
* ```
|
|
177
|
-
*/
|
|
178
|
-
sendRequest: (requestOverrides?: Partial<HttpRequestOptions>, optionsOverrides?: SendRequestOptions) => Promise<Response>;
|
|
179
|
-
/**
|
|
180
|
-
* Register a function to be called at the very end of the request handling process. This is useful for resources that should be accessible to error handlers, for instance.
|
|
181
|
-
*/
|
|
182
|
-
registerDeferredCleanup(cleanup: () => Promise<unknown>): void;
|
|
183
|
-
/**
|
|
184
|
-
* Gives the current request `secs` more seconds to finish, for when how long it needs is only apparent
|
|
185
|
-
* once it is already running - a listing page that turns out to have far more to scroll through than
|
|
186
|
-
* usual, say. Prefer `requestHandlerTimeoutSecs`, or a per-route override via
|
|
187
|
-
* {@link Router.addHandler|`router.addHandler`}, whenever the time needed is known up front.
|
|
188
|
-
*
|
|
189
|
-
* ```ts
|
|
190
|
-
* router.addHandler('LIST', async ({ extendTimeout, page }) => {
|
|
191
|
-
* const pageCount = await countPages(page);
|
|
192
|
-
* extendTimeout(pageCount * 10);
|
|
193
|
-
* await scrapeAllPages(page);
|
|
194
|
-
* });
|
|
195
|
-
* ```
|
|
196
|
-
*
|
|
197
|
-
* Extends the request handler's own timeout and the crawler's internal one together, so the extension
|
|
198
|
-
* is not immediately undone by the latter. Calling it from a handler that has already timed out does
|
|
199
|
-
* nothing.
|
|
200
|
-
*/
|
|
201
|
-
extendTimeout(secs: number): void;
|
|
202
|
-
}
|
|
203
|
-
/**
|
|
204
|
-
* A partial implementation of {@link RestrictedCrawlingContext} that stores parameters of calls to context methods for later inspection.
|
|
205
|
-
*
|
|
206
|
-
* @experimental
|
|
207
|
-
*/
|
|
208
|
-
export declare class RequestHandlerResult {
|
|
209
|
-
private configuration;
|
|
210
|
-
private crawleeStateKey;
|
|
211
|
-
private _keyValueStoreChanges;
|
|
212
|
-
private pushDataCalls;
|
|
213
|
-
private addRequestsCalls;
|
|
214
|
-
constructor(configuration: Configuration, crawleeStateKey: string);
|
|
215
|
-
/**
|
|
216
|
-
* A record of calls to {@link RestrictedCrawlingContext.pushData}, {@link RestrictedCrawlingContext.addRequests}, {@link RestrictedCrawlingContext.enqueueLinks} made by a request handler.
|
|
217
|
-
*/
|
|
218
|
-
get calls(): ReadonlyDeep<{
|
|
219
|
-
pushData: Parameters<RestrictedCrawlingContext['pushData']>[];
|
|
220
|
-
addRequests: Parameters<RestrictedCrawlingContext['addRequests']>[];
|
|
221
|
-
}>;
|
|
222
|
-
/**
|
|
223
|
-
* A record of changes made to key-value stores by a request handler.
|
|
224
|
-
*/
|
|
225
|
-
get keyValueStoreChanges(): ReadonlyDeep<Record<string, Record<string, {
|
|
226
|
-
changedValue: unknown;
|
|
227
|
-
options?: RecordOptions;
|
|
228
|
-
}>>>;
|
|
229
|
-
/**
|
|
230
|
-
* Items added to datasets by a request handler.
|
|
231
|
-
*/
|
|
232
|
-
get datasetItems(): ReadonlyDeep<{
|
|
233
|
-
item: Dictionary;
|
|
234
|
-
datasetIdentifier?: string | StorageIdentifier;
|
|
235
|
-
}[]>;
|
|
236
|
-
/**
|
|
237
|
-
* URLs enqueued to the request queue by a request handler, either via {@link RestrictedCrawlingContext.addRequests} or {@link RestrictedCrawlingContext.enqueueLinks}
|
|
238
|
-
*/
|
|
239
|
-
get enqueuedUrls(): ReadonlyDeep<{
|
|
240
|
-
url: string;
|
|
241
|
-
label?: string;
|
|
242
|
-
}[]>;
|
|
243
|
-
/**
|
|
244
|
-
* URL lists enqueued to the request queue by a request handler via {@link RestrictedCrawlingContext.addRequests} using the `requestsFromUrl` option.
|
|
245
|
-
*/
|
|
246
|
-
get enqueuedUrlLists(): ReadonlyDeep<{
|
|
247
|
-
listUrl: string;
|
|
248
|
-
label?: string;
|
|
249
|
-
}[]>;
|
|
250
|
-
pushData: RestrictedCrawlingContext['pushData'];
|
|
251
|
-
addRequests: RestrictedCrawlingContext['addRequests'];
|
|
252
|
-
useState: RestrictedCrawlingContext['useState'];
|
|
253
|
-
getKeyValueStore: RestrictedCrawlingContext['getKeyValueStore'];
|
|
254
|
-
private getKeyValueStoreChangedValue;
|
|
255
|
-
private setKeyValueStoreChangedValue;
|
|
256
|
-
}
|
|
257
|
-
export {};
|
|
@@ -1,107 +0,0 @@
|
|
|
1
|
-
import { KeyValueStore } from '../storages/key_value_store.js';
|
|
2
|
-
/**
|
|
3
|
-
* A partial implementation of {@link RestrictedCrawlingContext} that stores parameters of calls to context methods for later inspection.
|
|
4
|
-
*
|
|
5
|
-
* @experimental
|
|
6
|
-
*/
|
|
7
|
-
export class RequestHandlerResult {
|
|
8
|
-
configuration;
|
|
9
|
-
crawleeStateKey;
|
|
10
|
-
_keyValueStoreChanges = {};
|
|
11
|
-
pushDataCalls = [];
|
|
12
|
-
addRequestsCalls = [];
|
|
13
|
-
constructor(configuration, crawleeStateKey) {
|
|
14
|
-
this.configuration = configuration;
|
|
15
|
-
this.crawleeStateKey = crawleeStateKey;
|
|
16
|
-
}
|
|
17
|
-
/**
|
|
18
|
-
* A record of calls to {@link RestrictedCrawlingContext.pushData}, {@link RestrictedCrawlingContext.addRequests}, {@link RestrictedCrawlingContext.enqueueLinks} made by a request handler.
|
|
19
|
-
*/
|
|
20
|
-
get calls() {
|
|
21
|
-
return {
|
|
22
|
-
pushData: this.pushDataCalls,
|
|
23
|
-
addRequests: this.addRequestsCalls,
|
|
24
|
-
};
|
|
25
|
-
}
|
|
26
|
-
/**
|
|
27
|
-
* A record of changes made to key-value stores by a request handler.
|
|
28
|
-
*/
|
|
29
|
-
get keyValueStoreChanges() {
|
|
30
|
-
return this._keyValueStoreChanges;
|
|
31
|
-
}
|
|
32
|
-
/**
|
|
33
|
-
* Items added to datasets by a request handler.
|
|
34
|
-
*/
|
|
35
|
-
get datasetItems() {
|
|
36
|
-
return this.pushDataCalls.flatMap(([data, datasetIdentifier]) => (Array.isArray(data) ? data : [data]).map((item) => ({ item, datasetIdentifier })));
|
|
37
|
-
}
|
|
38
|
-
/**
|
|
39
|
-
* URLs enqueued to the request queue by a request handler, either via {@link RestrictedCrawlingContext.addRequests} or {@link RestrictedCrawlingContext.enqueueLinks}
|
|
40
|
-
*/
|
|
41
|
-
get enqueuedUrls() {
|
|
42
|
-
const result = [];
|
|
43
|
-
for (const [requests] of this.addRequestsCalls) {
|
|
44
|
-
for (const request of requests) {
|
|
45
|
-
if (typeof request === 'object' &&
|
|
46
|
-
(!('requestsFromUrl' in request) || request.requestsFromUrl !== undefined) &&
|
|
47
|
-
request.url !== undefined) {
|
|
48
|
-
result.push({ url: request.url, label: request.label });
|
|
49
|
-
}
|
|
50
|
-
else if (typeof request === 'string') {
|
|
51
|
-
result.push({ url: request });
|
|
52
|
-
}
|
|
53
|
-
}
|
|
54
|
-
}
|
|
55
|
-
return result;
|
|
56
|
-
}
|
|
57
|
-
/**
|
|
58
|
-
* URL lists enqueued to the request queue by a request handler via {@link RestrictedCrawlingContext.addRequests} using the `requestsFromUrl` option.
|
|
59
|
-
*/
|
|
60
|
-
get enqueuedUrlLists() {
|
|
61
|
-
const result = [];
|
|
62
|
-
for (const [requests] of this.addRequestsCalls) {
|
|
63
|
-
for (const request of requests) {
|
|
64
|
-
if (typeof request === 'object' &&
|
|
65
|
-
'requestsFromUrl' in request &&
|
|
66
|
-
request.requestsFromUrl !== undefined) {
|
|
67
|
-
result.push({ listUrl: request.requestsFromUrl, label: request.label });
|
|
68
|
-
}
|
|
69
|
-
}
|
|
70
|
-
}
|
|
71
|
-
return result;
|
|
72
|
-
}
|
|
73
|
-
pushData = async (data, datasetIdOrName) => {
|
|
74
|
-
this.pushDataCalls.push([data, datasetIdOrName]);
|
|
75
|
-
};
|
|
76
|
-
addRequests = async (requests, options = {}) => {
|
|
77
|
-
this.addRequestsCalls.push([requests, options]);
|
|
78
|
-
};
|
|
79
|
-
useState = async (defaultValue) => {
|
|
80
|
-
const store = await this.getKeyValueStore(undefined);
|
|
81
|
-
return await store.getAutoSavedValue(this.crawleeStateKey, defaultValue);
|
|
82
|
-
};
|
|
83
|
-
getKeyValueStore = async (identifier) => {
|
|
84
|
-
const store = await KeyValueStore.open(identifier, { configuration: this.configuration });
|
|
85
|
-
const storeId = store.id;
|
|
86
|
-
return {
|
|
87
|
-
id: storeId ?? this.configuration.defaultKeyValueStoreId,
|
|
88
|
-
name: store.name,
|
|
89
|
-
getValue: async (key) => this.getKeyValueStoreChangedValue(storeId, key) ?? (await store.getValue(key)),
|
|
90
|
-
setValue: async (key, value, options) => {
|
|
91
|
-
this.setKeyValueStoreChangedValue(storeId, key, value, options);
|
|
92
|
-
},
|
|
93
|
-
getAutoSavedValue: store.getAutoSavedValue.bind(store),
|
|
94
|
-
getPublicUrl: store.getPublicUrl.bind(store),
|
|
95
|
-
};
|
|
96
|
-
};
|
|
97
|
-
getKeyValueStoreChangedValue = (storeKey, key) => {
|
|
98
|
-
const id = storeKey ?? this.configuration.defaultKeyValueStoreId;
|
|
99
|
-
this._keyValueStoreChanges[id] ??= {};
|
|
100
|
-
return this.keyValueStoreChanges[id][key]?.changedValue ?? null;
|
|
101
|
-
};
|
|
102
|
-
setKeyValueStoreChangedValue = (storeKey, key, changedValue, options) => {
|
|
103
|
-
const id = storeKey ?? this.configuration.defaultKeyValueStoreId;
|
|
104
|
-
this._keyValueStoreChanges[id] ??= {};
|
|
105
|
-
this._keyValueStoreChanges[id][key] = { changedValue, options };
|
|
106
|
-
};
|
|
107
|
-
}
|
|
@@ -1,59 +0,0 @@
|
|
|
1
|
-
import type { CrawlingContext } from '../crawlers/crawler_commons.js';
|
|
2
|
-
import type { KeyValueStore } from '../storages/key_value_store.js';
|
|
3
|
-
import type { ErrnoException } from './error_tracker.js';
|
|
4
|
-
import type { SnapshottableProperties } from './internals/types.js';
|
|
5
|
-
interface BrowserCrawlingContext {
|
|
6
|
-
saveSnapshot: (options: {
|
|
7
|
-
key: string;
|
|
8
|
-
}) => Promise<void>;
|
|
9
|
-
}
|
|
10
|
-
export interface SnapshotResult {
|
|
11
|
-
screenshotFileName?: string;
|
|
12
|
-
htmlFileName?: string;
|
|
13
|
-
}
|
|
14
|
-
interface ErrorSnapshot {
|
|
15
|
-
screenshotFileName?: string;
|
|
16
|
-
screenshotFileUrl?: string;
|
|
17
|
-
htmlFileName?: string;
|
|
18
|
-
htmlFileUrl?: string;
|
|
19
|
-
}
|
|
20
|
-
/**
|
|
21
|
-
* ErrorSnapshotter class is used to capture a screenshot of the page and a snapshot of the HTML when an error occurs during web crawling.
|
|
22
|
-
*
|
|
23
|
-
* This functionality is opt-in, and can be enabled via the crawler options:
|
|
24
|
-
*
|
|
25
|
-
* ```ts
|
|
26
|
-
* const crawler = new BasicCrawler({
|
|
27
|
-
* // ...
|
|
28
|
-
* statisticsOptions: {
|
|
29
|
-
* saveErrorSnapshots: true,
|
|
30
|
-
* },
|
|
31
|
-
* });
|
|
32
|
-
* ```
|
|
33
|
-
*/
|
|
34
|
-
export declare class ErrorSnapshotter {
|
|
35
|
-
static readonly MAX_ERROR_CHARACTERS = 30;
|
|
36
|
-
static readonly MAX_HASH_LENGTH = 30;
|
|
37
|
-
static readonly MAX_FILENAME_LENGTH = 250;
|
|
38
|
-
static readonly BASE_MESSAGE = "An error occurred";
|
|
39
|
-
static readonly SNAPSHOT_PREFIX = "ERROR_SNAPSHOT";
|
|
40
|
-
/**
|
|
41
|
-
* Capture a snapshot of the error context.
|
|
42
|
-
*/
|
|
43
|
-
captureSnapshot(error: ErrnoException, context: CrawlingContext & SnapshottableProperties): Promise<ErrorSnapshot>;
|
|
44
|
-
/**
|
|
45
|
-
* Captures a snapshot of the current page using the context.saveSnapshot function.
|
|
46
|
-
* This function is applicable for browser contexts only.
|
|
47
|
-
* Returns an object containing the filenames of the screenshot and HTML file.
|
|
48
|
-
*/
|
|
49
|
-
contextCaptureSnapshot(context: BrowserCrawlingContext, fileName: string): Promise<SnapshotResult | undefined>;
|
|
50
|
-
/**
|
|
51
|
-
* Save the HTML snapshot of the page, and return the fileName with the extension.
|
|
52
|
-
*/
|
|
53
|
-
saveHTMLSnapshot(html: string, keyValueStore: Pick<KeyValueStore, 'setValue'>, fileName: string): Promise<string | undefined>;
|
|
54
|
-
/**
|
|
55
|
-
* Generate a unique fileName for each error snapshot.
|
|
56
|
-
*/
|
|
57
|
-
generateFilename(error: ErrnoException): string;
|
|
58
|
-
}
|
|
59
|
-
export {};
|
|
@@ -1,117 +0,0 @@
|
|
|
1
|
-
import crypto from 'node:crypto';
|
|
2
|
-
/**
|
|
3
|
-
* ErrorSnapshotter class is used to capture a screenshot of the page and a snapshot of the HTML when an error occurs during web crawling.
|
|
4
|
-
*
|
|
5
|
-
* This functionality is opt-in, and can be enabled via the crawler options:
|
|
6
|
-
*
|
|
7
|
-
* ```ts
|
|
8
|
-
* const crawler = new BasicCrawler({
|
|
9
|
-
* // ...
|
|
10
|
-
* statisticsOptions: {
|
|
11
|
-
* saveErrorSnapshots: true,
|
|
12
|
-
* },
|
|
13
|
-
* });
|
|
14
|
-
* ```
|
|
15
|
-
*/
|
|
16
|
-
export class ErrorSnapshotter {
|
|
17
|
-
static MAX_ERROR_CHARACTERS = 30;
|
|
18
|
-
static MAX_HASH_LENGTH = 30;
|
|
19
|
-
static MAX_FILENAME_LENGTH = 250;
|
|
20
|
-
static BASE_MESSAGE = 'An error occurred';
|
|
21
|
-
static SNAPSHOT_PREFIX = 'ERROR_SNAPSHOT';
|
|
22
|
-
/**
|
|
23
|
-
* Capture a snapshot of the error context.
|
|
24
|
-
*/
|
|
25
|
-
async captureSnapshot(error, context) {
|
|
26
|
-
try {
|
|
27
|
-
const page = context?.page;
|
|
28
|
-
const body = context?.body;
|
|
29
|
-
const keyValueStore = await context?.getKeyValueStore();
|
|
30
|
-
// If the key-value store is not available, or the body and page are not available, return empty filenames
|
|
31
|
-
if (!keyValueStore || (!body && !page)) {
|
|
32
|
-
return {};
|
|
33
|
-
}
|
|
34
|
-
const fileName = this.generateFilename(error);
|
|
35
|
-
let screenshotFileName;
|
|
36
|
-
let htmlFileName;
|
|
37
|
-
if (page) {
|
|
38
|
-
const capturedFiles = await this.contextCaptureSnapshot(context, fileName);
|
|
39
|
-
if (capturedFiles) {
|
|
40
|
-
screenshotFileName = capturedFiles.screenshotFileName;
|
|
41
|
-
htmlFileName = capturedFiles.htmlFileName;
|
|
42
|
-
}
|
|
43
|
-
// If the snapshot for browsers failed to capture the HTML, try to capture it from the page content
|
|
44
|
-
if (!htmlFileName) {
|
|
45
|
-
const html = await page.content();
|
|
46
|
-
htmlFileName = html ? await this.saveHTMLSnapshot(html, keyValueStore, fileName) : undefined;
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
else if (typeof body === 'string') {
|
|
50
|
-
// for non-browser contexts
|
|
51
|
-
htmlFileName = await this.saveHTMLSnapshot(body, keyValueStore, fileName);
|
|
52
|
-
}
|
|
53
|
-
return {
|
|
54
|
-
screenshotFileName,
|
|
55
|
-
screenshotFileUrl: screenshotFileName && (await keyValueStore.getPublicUrl(screenshotFileName)),
|
|
56
|
-
htmlFileName,
|
|
57
|
-
htmlFileUrl: htmlFileName && (await keyValueStore.getPublicUrl(htmlFileName)),
|
|
58
|
-
};
|
|
59
|
-
}
|
|
60
|
-
catch {
|
|
61
|
-
return {};
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
/**
|
|
65
|
-
* Captures a snapshot of the current page using the context.saveSnapshot function.
|
|
66
|
-
* This function is applicable for browser contexts only.
|
|
67
|
-
* Returns an object containing the filenames of the screenshot and HTML file.
|
|
68
|
-
*/
|
|
69
|
-
async contextCaptureSnapshot(context, fileName) {
|
|
70
|
-
try {
|
|
71
|
-
await context.saveSnapshot({ key: fileName });
|
|
72
|
-
return {
|
|
73
|
-
screenshotFileName: `${fileName}.jpg`,
|
|
74
|
-
htmlFileName: `${fileName}.html`,
|
|
75
|
-
};
|
|
76
|
-
}
|
|
77
|
-
catch {
|
|
78
|
-
return undefined;
|
|
79
|
-
}
|
|
80
|
-
}
|
|
81
|
-
/**
|
|
82
|
-
* Save the HTML snapshot of the page, and return the fileName with the extension.
|
|
83
|
-
*/
|
|
84
|
-
async saveHTMLSnapshot(html, keyValueStore, fileName) {
|
|
85
|
-
try {
|
|
86
|
-
await keyValueStore.setValue(fileName, html, { contentType: 'text/html' });
|
|
87
|
-
return `${fileName}.html`;
|
|
88
|
-
}
|
|
89
|
-
catch {
|
|
90
|
-
return undefined;
|
|
91
|
-
}
|
|
92
|
-
}
|
|
93
|
-
/**
|
|
94
|
-
* Generate a unique fileName for each error snapshot.
|
|
95
|
-
*/
|
|
96
|
-
generateFilename(error) {
|
|
97
|
-
const { SNAPSHOT_PREFIX, BASE_MESSAGE, MAX_HASH_LENGTH, MAX_ERROR_CHARACTERS, MAX_FILENAME_LENGTH } = ErrorSnapshotter;
|
|
98
|
-
// Create a hash of the error stack trace
|
|
99
|
-
const errorStackHash = crypto
|
|
100
|
-
.createHash('sha1')
|
|
101
|
-
.update(error.stack || error.message || '')
|
|
102
|
-
.digest('hex')
|
|
103
|
-
.slice(0, MAX_HASH_LENGTH);
|
|
104
|
-
const errorMessagePrefix = (error.message || BASE_MESSAGE).slice(0, MAX_ERROR_CHARACTERS).trim();
|
|
105
|
-
/**
|
|
106
|
-
* Remove non-word characters from the start and end of a string.
|
|
107
|
-
*/
|
|
108
|
-
const sanitizeString = (str) => {
|
|
109
|
-
return str.replace(/^\W+|\W+$/g, '');
|
|
110
|
-
};
|
|
111
|
-
// Generate fileName and remove disallowed characters
|
|
112
|
-
const fileName = `${SNAPSHOT_PREFIX}_${sanitizeString(errorStackHash)}_${sanitizeString(errorMessagePrefix)}`
|
|
113
|
-
.replace(/\W+/g, '-') // Replace non-word characters with a dash
|
|
114
|
-
.slice(0, MAX_FILENAME_LENGTH);
|
|
115
|
-
return fileName;
|
|
116
|
-
}
|
|
117
|
-
}
|
|
@@ -1,54 +0,0 @@
|
|
|
1
|
-
import type { CrawlingContext } from '../crawlers/crawler_commons.js';
|
|
2
|
-
import { ErrorSnapshotter } from './error_snapshotter.js';
|
|
3
|
-
import type { SnapshottableProperties } from './internals/types.js';
|
|
4
|
-
/**
|
|
5
|
-
* Node.js Error interface
|
|
6
|
-
*/
|
|
7
|
-
export interface ErrnoException extends Error {
|
|
8
|
-
errno?: number;
|
|
9
|
-
code?: string | number;
|
|
10
|
-
path?: string;
|
|
11
|
-
syscall?: string;
|
|
12
|
-
cause?: any;
|
|
13
|
-
}
|
|
14
|
-
export interface ErrorTrackerOptions {
|
|
15
|
-
showErrorCode: boolean;
|
|
16
|
-
showErrorName: boolean;
|
|
17
|
-
showStackTrace: boolean;
|
|
18
|
-
showFullStack: boolean;
|
|
19
|
-
showErrorMessage: boolean;
|
|
20
|
-
showFullMessage: boolean;
|
|
21
|
-
saveErrorSnapshots: boolean;
|
|
22
|
-
}
|
|
23
|
-
/**
|
|
24
|
-
* This class tracks errors and computes a summary of information like:
|
|
25
|
-
* - where the errors happened
|
|
26
|
-
* - what the error names are
|
|
27
|
-
* - what the error codes are
|
|
28
|
-
* - what is the general error message
|
|
29
|
-
*
|
|
30
|
-
* This is extremely useful when there are dynamic error messages, such as argument validation.
|
|
31
|
-
*
|
|
32
|
-
* Since the structure of the `tracker.result` object differs when using different options,
|
|
33
|
-
* it's typed as `Record<string, unknown>`. The most deep object has a `count` property, which is a number.
|
|
34
|
-
*
|
|
35
|
-
* It's possible to get the total amount of errors via the `tracker.total` property.
|
|
36
|
-
*/
|
|
37
|
-
export declare class ErrorTracker {
|
|
38
|
-
#private;
|
|
39
|
-
result: Record<string, unknown>;
|
|
40
|
-
total: number;
|
|
41
|
-
errorSnapshotter?: ErrorSnapshotter;
|
|
42
|
-
constructor(options?: Partial<ErrorTrackerOptions>);
|
|
43
|
-
private updateGroup;
|
|
44
|
-
add(error: ErrnoException): void;
|
|
45
|
-
/**
|
|
46
|
-
* This method is async, because it captures a snapshot of the error context.
|
|
47
|
-
* We added this new method to avoid breaking changes.
|
|
48
|
-
*/
|
|
49
|
-
addAsync(error: ErrnoException, context?: CrawlingContext): Promise<void>;
|
|
50
|
-
getUniqueErrorCount(): number;
|
|
51
|
-
getMostPopularErrors(count: number): [number, string[]][];
|
|
52
|
-
captureSnapshot(storage: Record<string, unknown>, error: ErrnoException, context: CrawlingContext & SnapshottableProperties): Promise<void>;
|
|
53
|
-
reset(): void;
|
|
54
|
-
}
|