@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +1 -1
  2. package/configuration.d.ts +16 -47
  3. package/configuration.js +13 -25
  4. package/debug.js +4 -4
  5. package/errors.d.ts +28 -38
  6. package/errors.js +33 -47
  7. package/events/event_manager.d.ts +2 -2
  8. package/events/event_manager.js +7 -6
  9. package/events/index.d.ts +1 -0
  10. package/events/local_event_manager.d.ts +1 -8
  11. package/events/local_event_manager.js +13 -13
  12. package/events/system_info.d.ts +38 -0
  13. package/index.d.ts +2 -8
  14. package/index.js +4 -8
  15. package/internal.d.ts +8 -0
  16. package/internal.js +9 -0
  17. package/log.d.ts +10 -11
  18. package/log.js +52 -20
  19. package/memory-storage/memory-storage.d.ts +15 -18
  20. package/memory-storage/memory-storage.js +80 -58
  21. package/memory-storage/resource-clients/dataset.d.ts +1 -6
  22. package/memory-storage/resource-clients/dataset.js +23 -31
  23. package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
  24. package/memory-storage/resource-clients/key-value-store.js +43 -67
  25. package/memory-storage/resource-clients/request-queue.d.ts +1 -42
  26. package/memory-storage/resource-clients/request-queue.js +109 -117
  27. package/owned_or_injected.d.ts +1 -3
  28. package/owned_or_injected.js +17 -17
  29. package/package.json +17 -20
  30. package/proxy_configuration.d.ts +21 -26
  31. package/proxy_configuration.js +35 -25
  32. package/recoverable_state.d.ts +104 -47
  33. package/recoverable_state.js +199 -74
  34. package/request.d.ts +20 -107
  35. package/request.js +78 -244
  36. package/serialization.js +17 -16
  37. package/service_locator.d.ts +22 -10
  38. package/service_locator.js +59 -48
  39. package/storages/batched_adds.d.ts +37 -0
  40. package/storages/batched_adds.js +73 -0
  41. package/storages/dataset.d.ts +13 -8
  42. package/storages/dataset.js +149 -40
  43. package/storages/index.d.ts +4 -4
  44. package/storages/index.js +2 -4
  45. package/storages/key_value_store.d.ts +16 -35
  46. package/storages/key_value_store.js +223 -110
  47. package/storages/key_value_store_codec.js +6 -11
  48. package/storages/request_dedup_cache.d.ts +1 -4
  49. package/storages/request_dedup_cache.js +15 -15
  50. package/storages/request_list.d.ts +9 -104
  51. package/storages/request_list.js +236 -233
  52. package/storages/request_loader.d.ts +49 -18
  53. package/storages/request_loader.js +36 -1
  54. package/storages/request_manager.d.ts +86 -0
  55. package/storages/request_manager_tandem.d.ts +14 -38
  56. package/storages/request_manager_tandem.js +67 -64
  57. package/storages/request_queue.d.ts +23 -50
  58. package/storages/request_queue.js +371 -226
  59. package/storages/storage_instance_manager.d.ts +2 -4
  60. package/storages/storage_instance_manager.js +21 -21
  61. package/storages/storage_stats.d.ts +1 -1
  62. package/storages/storage_stats.js +4 -4
  63. package/storages/transaction.d.ts +270 -0
  64. package/storages/transaction.js +296 -0
  65. package/storages/utils.d.ts +6 -3
  66. package/storages/utils.js +11 -2
  67. package/system-info/runtime.js +7 -7
  68. package/url.d.ts +9 -0
  69. package/url.js +11 -0
  70. package/validators.d.ts +23 -25
  71. package/validators.js +14 -25
  72. package/autoscaling/autoscaled_pool.d.ts +0 -213
  73. package/autoscaling/autoscaled_pool.js +0 -378
  74. package/autoscaling/client_load_signal.d.ts +0 -59
  75. package/autoscaling/client_load_signal.js +0 -73
  76. package/autoscaling/concurrency_system.d.ts +0 -283
  77. package/autoscaling/concurrency_system.js +0 -350
  78. package/autoscaling/cpu_load_signal.d.ts +0 -44
  79. package/autoscaling/cpu_load_signal.js +0 -46
  80. package/autoscaling/event_loop_load_signal.d.ts +0 -54
  81. package/autoscaling/event_loop_load_signal.js +0 -60
  82. package/autoscaling/index.d.ts +0 -9
  83. package/autoscaling/index.js +0 -9
  84. package/autoscaling/load_signal.d.ts +0 -99
  85. package/autoscaling/load_signal.js +0 -103
  86. package/autoscaling/memory_load_signal.d.ts +0 -56
  87. package/autoscaling/memory_load_signal.js +0 -106
  88. package/autoscaling/snapshotter.d.ts +0 -87
  89. package/autoscaling/snapshotter.js +0 -67
  90. package/autoscaling/system_status.d.ts +0 -161
  91. package/autoscaling/system_status.js +0 -139
  92. package/autoscaling/weighted_avg.d.ts +0 -5
  93. package/autoscaling/weighted_avg.js +0 -14
  94. package/cookie_utils.d.ts +0 -44
  95. package/cookie_utils.js +0 -122
  96. package/crawlers/context_pipeline.d.ts +0 -70
  97. package/crawlers/context_pipeline.js +0 -122
  98. package/crawlers/crawler_commons.d.ts +0 -257
  99. package/crawlers/crawler_commons.js +0 -107
  100. package/crawlers/error_snapshotter.d.ts +0 -59
  101. package/crawlers/error_snapshotter.js +0 -117
  102. package/crawlers/error_tracker.d.ts +0 -54
  103. package/crawlers/error_tracker.js +0 -308
  104. package/crawlers/index.d.ts +0 -5
  105. package/crawlers/index.js +0 -5
  106. package/crawlers/internals/types.d.ts +0 -7
  107. package/crawlers/statistics.d.ts +0 -209
  108. package/crawlers/statistics.js +0 -350
  109. package/enqueue_links/enqueue_links.d.ts +0 -264
  110. package/enqueue_links/enqueue_links.js +0 -271
  111. package/enqueue_links/index.d.ts +0 -2
  112. package/enqueue_links/index.js +0 -2
  113. package/enqueue_links/shared.d.ts +0 -83
  114. package/enqueue_links/shared.js +0 -221
  115. package/router.d.ts +0 -309
  116. package/router.js +0 -309
  117. package/session_pool/consts.d.ts +0 -3
  118. package/session_pool/consts.js +0 -3
  119. package/session_pool/errors.d.ts +0 -7
  120. package/session_pool/errors.js +0 -11
  121. package/session_pool/fingerprint.d.ts +0 -9
  122. package/session_pool/fingerprint.js +0 -30
  123. package/session_pool/index.d.ts +0 -4
  124. package/session_pool/index.js +0 -4
  125. package/session_pool/session.d.ts +0 -161
  126. package/session_pool/session.js +0 -218
  127. package/session_pool/session_pool.d.ts +0 -246
  128. package/session_pool/session_pool.js +0 -386
  129. package/storages/access_checking.d.ts +0 -12
  130. package/storages/access_checking.js +0 -17
  131. package/storages/sitemap_request_loader.d.ts +0 -249
  132. package/storages/sitemap_request_loader.js +0 -432
  133. /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
@@ -1,73 +0,0 @@
1
- import { betterClearInterval, betterSetInterval } from '@apify/utilities';
2
- import { serviceLocator } from '../service_locator.js';
3
- import { SnapshotStore } from './load_signal.js';
4
- const CLIENT_RATE_LIMIT_ERROR_RETRY_COUNT = 2;
5
- /**
6
- * Periodically checks the storage backend for rate-limit errors (HTTP 429) and reports overload when the error delta
7
- * exceeds a threshold.
8
- *
9
- * Built by default; construct one yourself only to wrap or adapt it — see {@link LoadSignal}.
10
- *
11
- * Switch it off entirely ({@link LoadSignalsOptions.client|`client: false`}) if the storage backend reports no
12
- * rate-limit statistics, since it otherwise polls it every second to no purpose.
13
- *
14
- * @category Scaling
15
- */
16
- export class ClientLoadSignal {
17
- name = 'clientInfo';
18
- overloadedRatio;
19
- store = new SnapshotStore();
20
- intervalMillis;
21
- maxErrors;
22
- interval;
23
- client;
24
- constructor(options = {}) {
25
- this.overloadedRatio = options.overloadedRatio ?? 0.3;
26
- this.intervalMillis = (options.snapshotIntervalSecs ?? 1) * 1000;
27
- this.maxErrors = options.maxErrors ?? 3;
28
- this.handle = this.handle.bind(this);
29
- }
30
- async start(context) {
31
- this.store.useSampleWindow(context.maxSampleWindowMillis);
32
- // A new session starts from a clean slate, or its first measurement diffs the error count against the previous
33
- // session's — possibly against a different backend, since the client is resolved afresh just below.
34
- this.store.clear();
35
- // Resolved here rather than in the constructor, where asking for the backend would instantiate a default one
36
- // as a side effect - long before the crawler that owns the run has had a chance to register its own.
37
- this.client = serviceLocator.getStorageBackend();
38
- this.interval = betterSetInterval(this.handle, this.intervalMillis);
39
- }
40
- async stop() {
41
- if (this.interval)
42
- betterClearInterval(this.interval);
43
- this.interval = undefined;
44
- this.client = undefined;
45
- }
46
- getSample(sampleDurationMillis) {
47
- return this.store.getSample(sampleDurationMillis);
48
- }
49
- /**
50
- * Records one snapshot, overloaded when rate-limit errors grew by more than the configured limit since the
51
- * previous one.
52
- * @internal Also lets tests drive the measurement without waiting on a timer.
53
- */
54
- handle(intervalCallback) {
55
- const now = new Date();
56
- const allErrorCounts = this.client?.stats?.rateLimitErrors ?? [];
57
- const currentErrCount = allErrorCounts[CLIENT_RATE_LIMIT_ERROR_RETRY_COUNT] || 0;
58
- const snapshot = {
59
- createdAt: now,
60
- isOverloaded: false,
61
- rateLimitErrorCount: currentErrCount,
62
- };
63
- const all = this.store.getAll();
64
- const previousSnapshot = all[all.length - 1];
65
- if (previousSnapshot) {
66
- const delta = currentErrCount - previousSnapshot.rateLimitErrorCount;
67
- if (delta > this.maxErrors)
68
- snapshot.isOverloaded = true;
69
- }
70
- this.store.push(snapshot, now);
71
- intervalCallback();
72
- }
73
- }
@@ -1,283 +0,0 @@
1
- import type { CrawleeLogger } from '../log.js';
2
- import type { LoadSignalsOptions } from './snapshotter.js';
3
- import type { SystemInfo } from './system_status.js';
4
- export interface ConcurrencySystemOptions {
5
- /**
6
- * The minimum number of tasks running in parallel.
7
- *
8
- * *WARNING:* If you set this value too high with respect to the available system memory and CPU, your code might run extremely slow or crash.
9
- * If you're not sure, just keep the default value and the concurrency will scale up automatically.
10
- * @default 1
11
- */
12
- minConcurrency?: number;
13
- /**
14
- * The maximum number of tasks running in parallel.
15
- * @default 200
16
- */
17
- maxConcurrency?: number;
18
- /**
19
- * The desired number of tasks that should be running parallel on the start of the pool,
20
- * if there is a large enough supply of them.
21
- * By default, it is `minConcurrency`.
22
- */
23
- desiredConcurrency?: number;
24
- /**
25
- * Minimum level of desired concurrency to reach before more scaling up is allowed.
26
- * @default 0.90
27
- */
28
- desiredConcurrencyRatio?: number;
29
- /**
30
- * Defines the fractional amount of desired concurrency to be added with each scaling up.
31
- * The minimum scaling step is one.
32
- * @default 0.05
33
- */
34
- scaleUpStepRatio?: number;
35
- /**
36
- * Defines the amount of desired concurrency to be subtracted with each scaling down.
37
- * The minimum scaling step is one.
38
- * @default 0.05
39
- */
40
- scaleDownStepRatio?: number;
41
- /**
42
- * Specifies a period in which the instance logs its state, in seconds.
43
- * Set to `null` to disable periodic logging.
44
- * @default 60
45
- */
46
- loggingIntervalSecs?: number | null;
47
- /**
48
- * Defines in seconds how often the system should attempt to adjust the desired concurrency
49
- * based on the latest system status. Setting it lower than 1 might have a severe impact on performance.
50
- * We suggest using a value from 5 to 20.
51
- * @default 10
52
- */
53
- autoscaleIntervalSecs?: number;
54
- /**
55
- * The signals that tell the system whether the machine is overloaded: per-resource tuning for the built-in four
56
- * (memory, event loop, CPU, client) plus any {@link LoadSignalsOptions.custom|`custom`} implementations of
57
- * your own. See {@link LoadSignalsOptions}.
58
- */
59
- loadSignals?: LoadSignalsOptions;
60
- /**
61
- * How far back the **autoscaling** decisions look, in seconds — the window the historical system status is
62
- * evaluated over, and therefore how much history the signals retain (the memory cost of raising it).
63
- * @default 30
64
- */
65
- snapshotHistorySecs?: number;
66
- /**
67
- * How far back the **task-gating** decision looks, in seconds — the window used to judge whether the system is
68
- * overloaded *right now*, before dispatching one more task. Deliberately shorter than
69
- * {@link ConcurrencySystemOptions.snapshotHistorySecs|`snapshotHistorySecs`}, so that dispatch reacts to
70
- * spikes quickly while scaling stays stable.
71
- * @default 5
72
- */
73
- currentHistorySecs?: number;
74
- /**
75
- * The maximum number of tasks per minute the system can run.
76
- * By default, this is set to `Infinity`, but you can pass any positive, non-zero integer.
77
- */
78
- maxTasksPerMinute?: number;
79
- log?: CrawleeLogger;
80
- }
81
- /**
82
- * Identifies *who* is asking a governor for capacity: one {@link AutoscaledPool}, or the crawler driving it. The
83
- * same object is passed on every call a pool makes, so per-consumer state can be keyed off it or off its `id`.
84
- * @category Scaling
85
- */
86
- export interface ConcurrencyConsumer {
87
- /** Process-unique and human-readable — a crawler's is its {@link BasicCrawlerOptions.id|`id`} option. */
88
- readonly id: string;
89
- }
90
- /**
91
- * The contract between an {@link AutoscaledPool} and its concurrency "governor" — the object that answers *is
92
- * there free compute for one more task?* and tracks the budget that tasks are booked against.
93
- * {@link ConcurrencySystem} is the canonical implementation; the interface lets alternate governors be substituted
94
- * without depending on its internals.
95
- *
96
- * Every allocation method is told which {@link ConcurrencyConsumer|consumer} is asking, so an implementation can
97
- * allocate per consumer. {@link ConcurrencySystem} does not: it serves whoever asks first, which can starve a pool
98
- * that joins a saturated system late.
99
- * @category Scaling
100
- */
101
- export interface IConcurrencySystem {
102
- /**
103
- * The number of tasks that should currently be running in parallel, assuming a sufficient supply of them. How it
104
- * is derived is up to the implementation, hence read-only here — but it must always be at least `1`, or a pool
105
- * could never start the first task.
106
- */
107
- readonly desiredConcurrency: number;
108
- /** The number of parallel tasks currently booked against this governor, regardless of which pool booked them. */
109
- readonly currentConcurrency: number;
110
- /**
111
- * Whether the governor is ready to be booked against. {@link AutoscaledPool.run|`pool.run()`} refuses to run
112
- * when this is `false`. An implementation with no startup lifecycle simply reports `true`.
113
- */
114
- readonly isRunning: boolean;
115
- /**
116
- * May **one more** task start right now, on behalf of `consumer`? A cheap pre-check the pool consults before
117
- * querying task readiness.
118
- *
119
- * Must **not** enforce rate limits that only make sense for ready tasks (e.g. a per-minute task cap): the pool
120
- * calls this before knowing whether any task is ready, so refusing here would stall an already-empty queue.
121
- *
122
- * Must also return `true` whenever `consumer` has nothing in flight of its own. A `false` sends that pool straight
123
- * to its finished-check **without** consulting `isTaskReadyFunction`, so a governor that starves an idle pool can
124
- * make its `run()` resolve while work is still pending. Tracking bookings per consumer answers that directly;
125
- * {@link ConcurrencySystem}, which does not, instead never refuses while
126
- * {@link IConcurrencySystem.currentConcurrency|`currentConcurrency`} is `0`.
127
- */
128
- hasCapacityForTask(consumer: ConcurrencyConsumer): boolean;
129
- /**
130
- * Books a task against the budget for `consumer`, returning `false` (without booking) when there is no room — the
131
- * budget is spent, the consumer is over its share, or an implementation-specific rate limit was reached.
132
- *
133
- * Must be an *atomic* (synchronous) check-and-book: several pools may share one governor, and a check separated
134
- * from the booking by an `await` lets two of them claim the last free slot at once.
135
- */
136
- tryRegisterTaskStart(consumer: ConcurrencyConsumer): boolean;
137
- /** Returns a task's slot to `consumer`'s budget. Called once the task settles (resolve or reject). */
138
- registerTaskEnd(consumer: ConcurrencyConsumer): void;
139
- }
140
- /**
141
- * The shareable "governor" behind an {@link AutoscaledPool}: it decides whether there is free compute for one more
142
- * task by combining live system load (via an internal {@link Snapshotter}) with a concurrency budget it autoscales
143
- * over time.
144
- *
145
- * Sharing one instance between several pools (and therefore several crawlers) caps their *combined* compute, instead
146
- * of letting each scale independently and oversubscribe the machine.
147
- *
148
- * Whoever builds the instance owns its lifecycle: call {@link ConcurrencySystem.start|`start()`} before any
149
- * borrowing pool runs and {@link ConcurrencySystem.stop|`stop()`} once they are all done (crawlers do this for the
150
- * default system they build, never for an injected one). Both calls are idempotent, and the first `stop()` tears the
151
- * system down for every borrower.
152
- * @category Scaling
153
- */
154
- export declare class ConcurrencySystem implements IConcurrencySystem {
155
- private readonly log;
156
- private readonly desiredConcurrencyRatio;
157
- private readonly scaleUpStepRatio;
158
- private readonly scaleDownStepRatio;
159
- private readonly loggingIntervalMillis;
160
- private readonly autoscaleIntervalMillis;
161
- private readonly maxTasksPerMinute;
162
- private _minConcurrency;
163
- private _maxConcurrency;
164
- private _desiredConcurrency;
165
- private _currentConcurrency;
166
- private lastLoggingTime?;
167
- private _tasksPerMinute;
168
- private readonly snapshotter;
169
- private readonly loadSignals;
170
- private readonly systemStatus;
171
- private autoscaleInterval?;
172
- private tasksDonePerSecondInterval?;
173
- /** Whether the snapshotter and autoscaling intervals are currently running. */
174
- private running;
175
- /** The in-flight (or completed) startup, memoized so concurrent `start()` calls await one boot. */
176
- private startPromise?;
177
- /** Set once per session, so a pool outliving `stop()` is reported once rather than every half second. */
178
- private warnedAboutQueryWhileStopped;
179
- constructor(options?: ConcurrencySystemOptions);
180
- /**
181
- * Gets the minimum number of tasks running in parallel.
182
- */
183
- get minConcurrency(): number;
184
- /**
185
- * Sets the minimum number of tasks running in parallel.
186
- *
187
- * *WARNING:* If you set this value too high with respect to the available system memory and CPU, your code might run extremely slow or crash.
188
- * If you're not sure, just keep the default value and the concurrency will scale up automatically.
189
- */
190
- set minConcurrency(value: number);
191
- /**
192
- * Gets the maximum number of tasks running in parallel.
193
- */
194
- get maxConcurrency(): number;
195
- /**
196
- * Sets the maximum number of tasks running in parallel. Lowering it below the current
197
- * {@link ConcurrencySystem.desiredConcurrency|`desiredConcurrency`} pulls that down to the new ceiling too, so
198
- * the change takes effect immediately (in-flight tasks are never cancelled — the budget simply drains to the new
199
- * limit as they settle).
200
- */
201
- set maxConcurrency(value: number);
202
- /**
203
- * Gets the desired concurrency for the system,
204
- * which is an estimated number of parallel tasks that the system can currently support.
205
- */
206
- get desiredConcurrency(): number;
207
- /**
208
- * Sets the desired concurrency for the system, i.e. the number of tasks that should be running
209
- * in parallel if there's large enough supply of tasks.
210
- */
211
- set desiredConcurrency(value: number);
212
- /**
213
- * Re-establishes `minConcurrency <= desiredConcurrency <= maxConcurrency` after any of the three is retuned.
214
- * Dispatch gates on the desired value alone, so one stranded above `maxConcurrency` would make the ceiling
215
- * meaningless. A contradictory pair (`minConcurrency > maxConcurrency`) resolves in favour of the maximum, since
216
- * that is the limit callers set in order to protect something.
217
- */
218
- private clampDesiredConcurrency;
219
- get currentConcurrency(): number;
220
- /** Whether the system is currently monitoring load and autoscaling the budget. */
221
- get isRunning(): boolean;
222
- /**
223
- * Boots the underlying snapshotter and the autoscaling interval. Idempotent, so a shared system isn't restarted
224
- * when handed to another consumer; concurrent callers await one startup. Rejects, leaving nothing running, if a
225
- * signal fails to start.
226
- */
227
- start(): Promise<void>;
228
- private boot;
229
- /**
230
- * Stops the snapshotter and intervals. Idempotent and safe to call even if the system was never started.
231
- */
232
- stop(): Promise<void>;
233
- private shutDown;
234
- /**
235
- * Reports, once per session, that capacity is being queried on a system that isn't running — a mistake nothing
236
- * else catches, since {@link AutoscaledPool.run|`pool.run()`} only checks
237
- * {@link ConcurrencySystem.isRunning|`isRunning`} on the way in. Both the overload verdict and
238
- * `desiredConcurrency` are frozen at that point, so the borrowing pool would otherwise just quietly mis-scale.
239
- */
240
- private warnIfNotRunning;
241
- /**
242
- * May **one more** task start right now? Returns `false` when the shared budget is spent (desired concurrency
243
- * reached) or when the machine is overloaded past `minConcurrency`.
244
- *
245
- * One budget for the whole machine, so the asking consumer is ignored — and therefore optional here, unlike in the
246
- * interface, letting the answer be queried directly.
247
- */
248
- hasCapacityForTask(_consumer?: ConcurrencyConsumer): boolean;
249
- /** Whether the per-minute task cap has been reached. */
250
- private get isOverMaxRequestLimit();
251
- /**
252
- * Atomically books a task against the shared budget: re-checks
253
- * {@link ConcurrencySystem.hasCapacityForTask|`hasCapacityForTask()`} plus the per-minute task cap and
254
- * increments the current concurrency in one synchronous step, returning `false` (without booking) when there is no
255
- * room. Call right before the task actually runs.
256
- *
257
- * The cap is enforced here rather than in the pre-check so that an empty queue never blocks the pool for a whole
258
- * extra minute.
259
- */
260
- tryRegisterTaskStart(consumer?: ConcurrencyConsumer): boolean;
261
- /** Returns a slot to the shared budget, whoever booked it. */
262
- registerTaskEnd(_consumer?: ConcurrencyConsumer): void;
263
- /**
264
- * What the system currently makes of the machine: the per-signal overload verdicts, evaluated over the
265
- * task-gating window, exactly as {@link ConcurrencySystem.hasCapacityForTask|`hasCapacityForTask()`} sees them.
266
- * The one public window into load monitoring — useful for answering *why* a crawl is not scaling up.
267
- */
268
- getCurrentStatus(): SystemInfo;
269
- /**
270
- * Evaluates the historical system status and scales the shared desired concurrency up or down accordingly. Driven
271
- * by the autoscaling interval started in {@link ConcurrencySystem.start|`start()`}.
272
- */
273
- private _autoscale;
274
- /**
275
- * Scales the system up by increasing the desired concurrency by the scaleUpStepRatio.
276
- */
277
- private _scaleUp;
278
- /**
279
- * Scales the system down by decreasing the desired concurrency by the scaleDownStepRatio.
280
- */
281
- private _scaleDown;
282
- private _incrementTasksDonePerSecond;
283
- }